diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index 3c934ac..8009fec 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -6,7 +6,7 @@ }, "metadata": { "description": "Tsuga toolkit for AI coding agents: one plugin with the tsuga CLI driver, live-platform investigation, dashboards, incident workflows, OpenTelemetry instrumentation, Collector, signal-choice, telemetry debug, and audit skills.", - "version": "0.8.4" + "version": "0.9.0" }, "plugins": [ { diff --git a/AGENTS.md b/AGENTS.md index 91ad156..ddb6309 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -144,7 +144,7 @@ CLI output values (log messages, span names, error text) are attacker-influenced ### Source-file reading -Code-reading skills (`signal-choice-advisor`, `tsuga-audit-telemetry-quality`, `otel-*`) may read source files in the user's project, but: +Code-reading skills (`signal-choice-advisor`, `tsuga-audit`, `otel-*`) may read source files in the user's project, but: - Never read `.env`, `*.secret`, `*credentials*`, `*token*` — flag and stop - Never reproduce API keys, ingestion keys, or endpoint URLs found in source @@ -183,7 +183,7 @@ These govern how skills _behave when executing_ (distinct from the authoring/rep ### Addendum: instrumentation-quality skills -These additional rules apply to audit and design skills (`signal-choice-advisor`, `tsuga-audit-telemetry-quality`, `otel-*`). They extend — but do not replace — the 10 rules above. +These additional rules apply to audit and design skills (`signal-choice-advisor`, `tsuga-audit`, `otel-*`). They extend — but do not replace — the 10 rules above. **A1. Code reading is allowed and expected.** Audit and design skills may read source files in the user's project. CLI evidence tells you what arrived in Tsuga; code evidence tells you why. Both are valid. Neither is sufficient alone. diff --git a/plugins/tsuga/.claude-plugin/plugin.json b/plugins/tsuga/.claude-plugin/plugin.json index 3c07eea..0a0b768 100644 --- a/plugins/tsuga/.claude-plugin/plugin.json +++ b/plugins/tsuga/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "tsuga", "description": "Tsuga observability plugin: the `tsuga` CLI driver (commands, TQL syntax, aggregation bodies, counter math, deep links, cloud/k8s translators); live-platform investigation for service health, errors, latency, and monitor coverage; dashboard building; incident orchestration; OpenTelemetry SDK, Collector, OTTL, signal-choice, telemetry debug, and audit skills; and meta-skills for building and validating skill bundles.", - "version": "0.8.4", + "version": "0.9.0", "author": { "name": "Tsuga Engineering", "email": "engineering@tsuga.com" diff --git a/plugins/tsuga/skills/signal-choice-advisor/SKILL.md b/plugins/tsuga/skills/signal-choice-advisor/SKILL.md index 36e71f9..d920155 100644 --- a/plugins/tsuga/skills/signal-choice-advisor/SKILL.md +++ b/plugins/tsuga/skills/signal-choice-advisor/SKILL.md @@ -110,7 +110,7 @@ Estimate metric series count by multiplying unique values across all dimensions. - `otel-instrumentation` - SDK implementation after the signal and naming decision is made. - `otel-collector` - Collector transforms, filters, routing, redaction, and OTTL. -- `tsuga-audit-telemetry-quality` - audit existing metric design and broader telemetry quality issues. +- `tsuga-audit` - audit existing metric design and broader telemetry quality issues. - `tsuga-debug-telemetry-ingestion` - verify the signal arrives after implementation. ## Safety Rules diff --git a/plugins/tsuga/skills/tsuga-analyze-trace-latency/SKILL.md b/plugins/tsuga/skills/tsuga-analyze-trace-latency/SKILL.md index a457bb6..7446932 100644 --- a/plugins/tsuga/skills/tsuga-analyze-trace-latency/SKILL.md +++ b/plugins/tsuga/skills/tsuga-analyze-trace-latency/SKILL.md @@ -126,4 +126,4 @@ Trace-log correlation: matching traces found via trace_id / not attempted (s - `tsuga-investigate-service-health` — broader health triage including logs and metrics - `tsuga-investigate-errors` — error deep-dive if latency correlates with errors - `tsuga-debug-telemetry-ingestion` — verify traces are arriving if no spans found -- `tsuga-audit-telemetry-quality` — audit span design quality +- `tsuga-audit` — audit span design quality diff --git a/plugins/tsuga/skills/tsuga-audit/SKILL.md b/plugins/tsuga/skills/tsuga-audit/SKILL.md new file mode 100644 index 0000000..d95ba99 --- /dev/null +++ b/plugins/tsuga/skills/tsuga-audit/SKILL.md @@ -0,0 +1,103 @@ +--- +name: tsuga-audit +description: "Use for Tsuga setup audits, including quality reports, telemetry quality, alerting coverage, dashboard hygiene, routing, ingestion keys, tag policies, and retention policies. For broad or unclear audit requests, ask which audit class to run." +--- + +# Tsuga Audit + +Front door for auditing a Tsuga setup. This skill itself only decides which audit class (or classes) to run — the actual workflow, evidence rules, safety rules, and output format for each class live in `references/`, and each reference is a complete, self-contained procedure on its own. + +## Audit Classes + +| Class | Covers | Reference | +|---|---|---| +| Telemetry quality | Log structure/correlation, metric naming/units/temporality/cardinality, span naming/status/kind/noisy spans, quality reports, resource drift, high-cardinality attributes | `references/telemetry-quality.md` | +| Monitor & alerting coverage | Services without monitors, notification routing, silences, stale team references, PagerDuty/Slack destinations, coverage percentages | `references/monitor-coverage.md` | +| Dashboard hygiene | Empty/stale/unused dashboards, dashboard ownership, widget counts, teams without a dashboard | `references/dashboards.md` | +| Telemetry routing | The `routes` resource: unrouted logs, over-routed logs, route ownership/ranking, teams without a route | `references/routes.md` | +| Resource governance | Unused ingestion API keys, tag policy compliance, retention policy coverage/outliers | `references/resource-governance.md` | + +## Workflow + +1. **Pull the quality report first — it's free, pre-computed evidence spanning most of the classes below.** Quality reports are generated snapshots that score rules across logs, metrics, traces, resources, monitors, dashboards, routes, and ingestion keys in one pass; several rules map directly onto the classes below (resource governance's tag/retention-policy checks are the exception — quality reports don't score those yet, so that reference gets no seed evidence from this step). Pulling this before classifying or asking anything focuses the rest of the audit instead of starting from zero. + + ```bash + tsuga quality-reports list --team --rationale "..." + ``` + + - If the command errors on cluster selection, report the exact error and check `tsuga quality-reports list --help`; do not invent flags not shown by the installed CLI. + - Omit `--team` only when nothing is scoped yet (the vague-request case in step 3) — a team-scoped call omits global rows, so prefer narrow when a team or service is already known, consistent with the narrow-before-broad rule everywhere else in this plugin. + - Each row is one rule result: `ruleId`, `status` (`passed` / `failed` / `ignored`), `score`, `weight`, `owner` (team ID), `createdAt`, `recommendation`. `reportOverallScore` and `reportTotalWeight` repeat across every row from the same `reportId` — read them once, don't re-derive per row. + - Treat `min(rows.createdAt)` as the report generation time; flag it if older than 48 hours. This is a stored snapshot, not a live view — say so in output. + - `status: ignored` means a human explicitly suppressed that rule. Report it separately from `failed`; never fold it into a pass count. + - **Carry each row's `recommendation` text into the reference's Recommended Actions close to verbatim.** It's computed server-side per team/rule and is already more specific than anything worth re-deriving — exact attribute names, exact Collector processor names, exact bad values found (e.g. `user_id` → `user.id`), team-scoped counts. Paraphrasing it into generic advice throws away precision that's already correct. + - **Prioritize failing rows by estimated impact, not by list order.** The product UI shows "Estimated impact" (approximate team-score lift from fixing one rule) but doesn't expose it as an API/CLI field. Since a failed row scores `0`, approximate it yourself: `weight / reportTotalWeight` ≈ the score improvement from fixing that one rule. Treat this as directional — it isn't confirmed to match the UI's exact internal formula — but it's enough to answer "which one should I fix first." + - If the user is asking what a quality report *is* or how scoring works, rather than requesting an audit, fetch `tsuga docs get account-and-settings/quality-reports` for the concept. Don't fetch it for remediation guidance — the `recommendation` field per row already covers that, fresher and more specifically than the doc's prose. + + Route `failed` (and `ignored`, reported separately) rows to a class by `ruleId`: + + | `ruleId` family | Class | + |---|---| + | `no-empty-dashboards`, `no-stale-dashboards`, `no-unused-dashboards` | Dashboard hygiene | + | `monitor-has-notification`, `no-orphan-monitors`, `no-redundant-monitors`, `no-flapping-monitors`, `no-noisy-group-monitors`, `no-long-alert-monitors` | Monitor & alerting coverage | + | `team-has-route`, `unrouted-logs`, `no-over-routed-logs` | Telemetry routing — these are about the `routes` resource (who owns incoming data), not about monitors or notification-rules; don't lump them into monitor & alerting coverage even though the word "routing" is shared with notification routing. | + | `service-name-present`, `host-name-present`, `metric-usage`, `metric-unit-consistency`, `no-absent-log-fields`, `no-error-logs-as-info`, `no-debug-logs-in-prod`, `no-duplicate-attr-values`, `no-multiline-logs`, `no-future-dated-logs`, `no-orphan-spans`, `inconsistent-attribute-naming`, `use-standard-attributes`, `k8s-*`, `db-*`, `otel-collector-self-metrics` | Telemetry quality | + | `no-unused-ingestion-api-keys` | Resource governance | + | `no-unused-operation-api-keys`, or anything else that doesn't match a class above | Not owned by a reference yet — no CLI resource exists to manage or list operation API keys, so this stays report-only. Report it directly in a top-level "Other Quality Report Findings" note rather than forcing it into a class. | + + This list reflects the rule catalog observed at authoring time — quality-report rules evolve, so if an unfamiliar `ruleId` shows up, read its `recommendation` text to classify it rather than assuming it doesn't map to anything. + +2. **Classify the request.** Check it against the keyword lists in the Audit Classes table (and this skill's own description). If it clearly names one class, skip straight to step 4 for that reference. + +3. **If the request is broad or doesn't indicate a class** — e.g. "audit my tsuga setup," "audit everything," "how healthy is my Tsuga config" — ask before proceeding, using the failing-row counts from step 1 to make the choice concrete instead of abstract: + + > Which would you like audited? (Quality report shows N failing telemetry rules, M failing monitor rules, K failing dashboard rules, R failing routing rules, G failing governance rules for this scope.) + > 1. **Telemetry quality** — logs/metrics/traces shape, naming, cardinality, correlation + > 2. **Monitor & alerting coverage** — which services have monitors, notification routing, silences + > 3. **Dashboard hygiene** — empty, stale, or unused dashboards, dashboard ownership + > 4. **Telemetry routing** — unrouted or over-routed logs, route ownership + > 5. **Resource governance** — unused ingestion keys, tag policy compliance, retention coverage + > 6. **Everything / full sweep** + + Wait for the answer. Don't guess at scope beyond what's already in the request — the reference workflows each have their own required-input prompts for service/team/window, so only resolve the *class* here. + +4. **Load the matching reference file(s) and follow them exactly**, handing each the quality-report rows from step 1 that route to its class as seed evidence. Each reference owns its full procedure — required inputs, docs lookups, evidence rules, safety/mutation gates, output template. Do not summarize, skip, or re-derive steps from memory instead of reading the reference; they encode CLI quirks and evidence requirements that aren't obvious from the class name alone. A reference should still independently corroborate a quality-report finding with its own CLI evidence before presenting it as confirmed — except where a reference explicitly says a rule has no CLI equivalent (e.g. dashboard view counts), in which case label it as report-only evidence. + +5. **Full sweep** (user picks "everything"): run each reference's workflow independently against the same scope (same service/team/window where applicable), then present one combined report: + + ```markdown + # Combined Tsuga Audit + + ## Cross-Class Observations + + + ## Telemetry Quality + + + ## Monitor & Alerting Coverage + + + ## Dashboard Hygiene + + + ## Telemetry Routing + + + ## Resource Governance + + + ## Other Quality Report Findings + + ``` + + If one class's workflow needs input the others don't (e.g. a time window is meaningful for telemetry quality but optional for resource governance), resolve each independently rather than forcing a single set of inputs on all five. + +## Related Skills / Next Steps + +- `tsuga-cli` — underlying CLI syntax every reference depends on. +- `tsuga-debug-telemetry-ingestion` — missing/sparse telemetry or broken propagation; run before a telemetry-quality audit if data isn't arriving at all. +- `otel-instrumentation` — apply confirmed-language SDK/span/metric fixes after a telemetry-quality audit identifies a fix surface. +- `otel-collector` — fix Collector transforms, routing, redaction, or enrichment affecting signal quality. +- `signal-choice-advisor` — redesign signal choice, semantic names, or high-cardinality attributes found by a telemetry-quality audit. +- `tsuga-investigate-service-health` — check current health when a coverage audit finds a gap. +- `tsuga-build-dashboard` — fix, rebuild, or repopulate a dashboard flagged by a dashboard-hygiene audit. diff --git a/plugins/tsuga/skills/tsuga-audit/references/dashboards.md b/plugins/tsuga/skills/tsuga-audit/references/dashboards.md new file mode 100644 index 0000000..ea1ebd0 --- /dev/null +++ b/plugins/tsuga/skills/tsuga-audit/references/dashboards.md @@ -0,0 +1,107 @@ +# Dashboard Audit + +## Example Requests + +- "Audit our dashboards" +- "Which dashboards are empty or stale?" +- "Do we have unused dashboards we should clean up?" +- "Review dashboard hygiene for team X" +- "Does every team have a dashboard?" + +## Required Inputs + +- **Scope** (optional, default: all dashboards): can be narrowed to a specific team or dashboard. If scoping to all dashboards and the org has more than ~50, warn before proceeding — dashboards are usually far fewer than services, so this is a lower bar than the 100-service threshold used elsewhere. + +## Runtime Docs Lookup + +For docs lookup rationale and docs-error behavior, follow `tsuga-cli`; examples omit `--rationale` for brevity. + +| Need | Fetch | +|---|---| +| Dashboards product docs | `tsuga docs get visualize/dashboards/index` | +| Dashboard query/filter/sort fields | `tsuga docs get api/queryDashboards` | +| Quality report concepts and rule families | `tsuga docs get account-and-settings/quality-reports` | + +## Workflow + +1. Resolve requested team/dashboard scope. `tsuga dashboards list` supports server-side filtering by `owners`, `tags`, and `folderId` — use `-d ''` rather than filtering locally when scope is known. + +2. **Start from the quality report, not from scratch.** If this reference is running inside a `tsuga-audit` orchestrator pass, use the rows it already fetched. If invoked standalone, pull them directly: + ```bash + tsuga quality-reports list --team --rationale "..." + ``` + Filter to `ruleId` in `no-empty-dashboards`, `no-stale-dashboards`, `no-unused-dashboards`. These three are dashboard-hygiene rules scored on every report run — they tell you where to look before you run a single dashboard query. + - **`no-unused-dashboards` is quality-report-only evidence.** Dashboard view counts are not exposed through `tsuga dashboards list`/`get` or any documented API field — the quality report is the only source for "zero views in N days." Report it as `source: quality report (not independently verifiable via CLI)`, not as a CLI-confirmed finding. + - `no-empty-dashboards` and `no-stale-dashboards` *are* independently checkable — corroborate them in the next two steps rather than taking the report's pass/fail at face value, since the report is a stored snapshot and dashboards change after it was generated. + +3. `tsuga dashboards list -d ''` with `sort: {"by": "widgetCount", "direction": "asc"}` — confirm actual empty/near-empty dashboards. Neither `list` nor `get` returns a `widgetCount` scalar field (it's a sort key only); count widgets from each returned dashboard's `graphs` array length (`graphs.length` ≤ 1) rather than citing a field that isn't in the payload. + +4. Same call with `sort: {"by": "updatedAt", "direction": "asc"}` orders dashboards oldest-to-newest, but `updatedAt` is a sort key only — it's never returned as a field on the dashboard object in either `list` or `get`, so an exact "days since update" cannot be computed from CLI/API data alone. Use the sort order as directional corroboration (which dashboards rank oldest), and treat the quality report's `no-stale-dashboards` row — which has the actual staleness verdict and threshold, currently 180 days at time of writing — as the source of truth for the finding and threshold; read the live `recommendation` text rather than hardcoding the number. + +5. Resolve ownership: cross dashboard `owner` values against `tsuga teams list`. A dashboard owned by a team ID absent from `teams list` is a stale team reference, same pattern as monitor coverage. + +6. Team dashboard coverage: cross `tsuga teams list` against the distinct `owner` values returned by `dashboards list`. A team with zero dashboards is a coverage gap — surface it, but don't assume it's wrong; some teams legitimately rely on another team's shared dashboard. + +7. Cross-reference monitors when available: a quality-report `no-orphan-monitors` failure ("monitors not linked to a dashboard") for the same team corroborates a dashboard coverage gap — cite it as corroboration, don't re-derive monitor-linkage logic here; that check belongs to the monitor-coverage reference. + +8. Before recommending deletion or archival of any flagged dashboard, run `tsuga dashboards get ` to inspect its actual widgets. A quality-report snapshot can be stale — confirm the dashboard is still empty/untouched at the time of the audit, not just at report-generation time. + +## Evidence Rules + +- Every finding cites the command and value that produced it — quality-report rows cite `ruleId` + `createdAt` + `recommendation`; CLI findings cite the exact field value (`owner`) or derived value (`graphs.length` for widget count). `widgetCount` and `updatedAt` are sort keys, not returned fields — never cite them as if the API handed back that literal value. +- Label evidence as `source: tsuga CLI` or `source: quality report`. Findings that combine both should say so explicitly. +- Treat `min(rows.createdAt)` across quality-report rows as the report generation time; flag it if older than 48 hours, same convention used elsewhere. +- `status: ignored` rows are a human-suppressed result, not a pass — report them separately from `failed`, never silently drop or count them as healthy. +- Do not recommend deleting or archiving a dashboard from the quality-report score alone; confirm with `dashboards get` first (step 8). +- Carry each row's `recommendation` text into Recommended Actions close to verbatim, and prioritize multiple findings by estimated impact — see `tsuga-audit`'s Quality Reports step for the exact rule. + +## Safety Rules + +- Read-only by default. Archiving or deleting a dashboard requires the same mutation gate as everywhere else: show the proposed change and why, wait for explicit confirmation, apply only after. +- Never batch-delete or batch-archive dashboards without the user reviewing the full proposed list. +- State the quality-report query timestamp in output — it is a configuration/usage snapshot, not a live view. + +## Output Template + +```markdown +## Dashboard Audit +Scope: > | As of: | Quality report generated: + +## Summary +Dashboards audited: | Empty (widgetCount ≤ 1): | Stale (not updated in days): | Unused (quality report only): +Teams with zero dashboards: + +## Empty Dashboards +| Dashboard | Owner | Widget count | +|---|---|---| + +## Stale Dashboards +| Dashboard | Owner | Staleness evidence | +|---|---|---| + +## Unused Dashboards (quality report only — not independently verifiable via CLI) +| Dashboard | Owner | Recommendation text | +|---|---|---| + +## Teams Without Any Dashboard +| Team | Note | +|---|---| + +## Findings +| Finding | Evidence | Source | Confidence | +|---|---|---|---| + +## Recommended Actions + +## Limitations +- Dashboard view counts are not exposed via CLI/API; "unused" relies entirely on the quality report snapshot. +- `updatedAt` is documented as a dashboard query sort key; if it is not returned in the CLI payload, exact last-updated timestamps and days-since-update must come from the quality report recommendation, not CLI evidence. +- Widget-level query health (references to removed metrics or log fields) requires manual inspection via `tsuga dashboards get `, not checked automatically here. +- Stale-dashboard threshold is a rule parameter, not a fixed product guarantee — read it from the live `recommendation` text. +``` + +## Related Skills / Next Steps + +- `tsuga-build-dashboard` — fix, rebuild, or repopulate a flagged dashboard. +- `tsuga-cli` — resource command syntax, filter/sort bodies, owner lookups. +- `tsuga-audit` (monitor coverage class) — cross-check monitor-to-dashboard linkage when a coverage gap spans both. diff --git a/plugins/tsuga/skills/tsuga-audit-monitor-coverage/SKILL.md b/plugins/tsuga/skills/tsuga-audit/references/monitor-coverage.md similarity index 64% rename from plugins/tsuga/skills/tsuga-audit-monitor-coverage/SKILL.md rename to plugins/tsuga/skills/tsuga-audit/references/monitor-coverage.md index 84e11c1..76b1d1b 100644 --- a/plugins/tsuga/skills/tsuga-audit-monitor-coverage/SKILL.md +++ b/plugins/tsuga/skills/tsuga-audit/references/monitor-coverage.md @@ -1,9 +1,4 @@ ---- -name: tsuga-audit-monitor-coverage -description: "Use when asked to check monitor coverage, services without monitors, alerting gaps, notification routing, notification-rules, silences, stale team references, PagerDuty or Slack routing destinations, teams without configured alerts, monitor ownership, monitor filters, log-error-pattern coverage, active/inactive routing rules, coverage summaries, routing gaps, coverage percentages, or whether alert configuration covers a service/team scope." ---- - -# Audit Monitor Coverage +# Monitor Coverage Audit ## Example Requests @@ -18,25 +13,44 @@ description: "Use when asked to check monitor coverage, services without monitor - **Scope** (optional, default: all services): can be narrowed to a specific team or service. If scoping to all services, warn if the list exceeds 100 services before proceeding. +## Runtime Docs Lookup + +For docs lookup rationale and docs-error behavior, follow `tsuga-cli`; examples omit `--rationale` for brevity. + +Fetch before interpreting notification-rule matches or monitor snooze/cluster scoping: + +| Need | Fetch | +|---|---| +| Notification rule matching semantics (team/priority/status/cluster filters, additional-filter query subset) | `tsuga docs get alert/notifications/rules` | +| Monitor fields, snooze status, cluster scoping | `tsuga docs get alert/monitors/index` | + ## Workflow -1. Resolve requested service/team/env scope first. `tsuga services list` has no filter flags, so filter returned rows locally; if a full all-service audit would exceed 100 services, confirm scope with the user before continuing. +1. Start from the quality report. If this reference is running inside a `tsuga-audit` orchestrator pass, use the rows it already fetched. If invoked standalone, pull them directly: + ```bash + tsuga quality-reports list --team --rationale "..." + ``` + Filter to `ruleId` in `monitor-has-notification`, `no-orphan-monitors`, `no-redundant-monitors`, `no-flapping-monitors`, `no-noisy-group-monitors`, `no-long-alert-monitors`. These score every report run and tell you which teams/monitors to look at before running a single monitor query. -2. `tsuga monitors list` — monitor definitions. Use `-d ''` when a read-only server-side filter is available; otherwise filter locally. Build coverage using the same shapes the app uses for service-related resources: +2. Resolve requested service/team/env scope first. `tsuga services list` has no filter flags, so filter returned rows locally; if a full all-service audit would exceed 100 services, confirm scope with the user before continuing. + +3. `tsuga monitors list` — monitor definitions. Use `-d ''` when a read-only server-side filter is available; otherwise filter locally. Build coverage using the same shapes the app uses for service-related resources: - Aggregation monitors: parse `configuration.queries[].filter` for exact or glob `service:` and `context.service.name:` values, including quoted values. - Log-error-pattern monitors: check `configuration.filter.service`, `env`, and `teamIds` when present. - Deployment/cluster-scoped monitors: treat env/namespace/cluster matches as possible coverage and explain the match basis. + - Snoozed monitors: also run `tsuga monitors list -d '{"filters":{"activity":"snoozed"}}'`. A snoozed monitor is a saved definition, not an evaluating one — exclude it from the "with monitors (exact match)" coverage count and list it separately as not currently evaluating. -3. `tsuga teams list` — all teams; build `{team-id → team-name}` map. +4. `tsuga teams list` — all teams; build `{team-id → team-name}` map. -4. `tsuga notification-rules list` — evaluate active rules by CLI-visible matcher fields: `teamsFilter`, `prioritiesFilter`, `transitionTypesFilter`, `clusterIdsFilter`, `isActive`, and optional `queryString` when present. Treat `targets` as delivery destinations, not match constraints; label tag/dimension matching unverified unless `queryString` exposes it. +5. `tsuga notification-rules list` — evaluate active rules by CLI-visible matcher fields: `teamsFilter`, `prioritiesFilter`, `transitionTypesFilter`, `clusterIdsFilter`, `isActive`, and optional `queryString` when present. Per `alert/notifications/rules`: an empty `prioritiesFilter`/`transitionTypesFilter`/`clusterIdsFilter` matches all values on that dimension, and `clusterIdsFilter` never excludes a monitor with no cluster restriction — do not flag a cluster-less monitor as a routing gap solely because a rule's `clusterIdsFilter` is non-empty. `queryString`, when present, matches monitor tags/group-by dimensions with a restricted query subset (`key:value` plus uppercase `AND`/`OR`/`NOT` only — no ranges, prefixes, suffixes, or contains matches). Treat `targets` as delivery destinations, not match constraints. -5. `tsuga notification-silences list` — list active silences; note coverage scope and schedule type. For one-time silences report `endTime`; for recurring weekly silences report schedule and timezone. +6. `tsuga notification-silences list` — list active silences; note coverage scope and schedule type. For one-time silences report `endTime`; for recurring weekly silences report schedule and timezone. -6. Cross-reference: +7. Cross-reference: - Services not covered by any exact-match or supported monitor association → coverage gap - Monitor owner/team with no active matching notification rule after applying CLI-visible filters → routing gap - Notification rule `teamsFilter.teams[]` referencing team IDs not in `teams list` results → stale team reference + - Service covered only by a snoozed monitor → report as "not currently evaluating," not as covered ### Confirm Before Applying @@ -51,13 +65,15 @@ After deploy, recommend running `tsuga-debug-telemetry-ingestion` to verify sign ## Evidence Requirements - "No monitor coverage" = service name not found in exact `service:` / `context.service.name:` aggregation filters, log-error-pattern service filters, or app-supported service associations. Glob, env, namespace, tag, or cluster matches are listed separately as "possible or indirect coverage." -- "Routing gap" = no active notification rule matches the monitor/team after applying CLI-visible filters; target presence only proves a destination exists. +- "Routing gap" = no active notification rule matches the monitor/team after applying CLI-visible filters, accounting for `clusterIdsFilter`'s no-cluster-exclusion behavior; target presence only proves a destination exists. +- A monitor returned by `filters.activity: snoozed` counts toward "Snoozed" only, never toward "with monitors (exact match)." - Every finding cites the command and value that produced it. - State query timestamp in output. +- Quality-report findings: carry the row's `recommendation` text into Recommended Actions close to verbatim, and prioritize multiple findings by estimated impact — see `tsuga-audit`'s Quality Reports step for the exact rule. ## Output Template -``` +```markdown ## Monitor Coverage Audit Scope: / service > | As of: @@ -65,12 +81,18 @@ Scope: / service > | As of: Services audited: | With monitors (exact match): (%) | No monitors: Teams with monitors but no active notification rule: Active silences: +Snoozed monitors (not currently evaluating): ## Uncovered Services (no exact or supported monitor association) | Service | Team | Env | |---|---|---| | | | | +## Snoozed Monitors (not currently evaluating) +| Monitor | Service/Team | Note | +|---|---|---| +| | | Snoozed — excluded from coverage count above | + ## Possible or Indirect Coverage The following monitor filters use glob, env, namespace, tag, or cluster scope and may cover services above: - : filter pattern (owner: ) @@ -99,13 +121,13 @@ tsuga monitors create -d '' # Create a notification rule for tsuga notification-rules create -d '' ``` +``` ## Limitations - Monitor coverage uses known monitor associations from config fields; unsupported custom filters may still need manual review - Services are telemetry-derived inventory snapshots, not an authoritative service ownership registry - Monitor firing state not available (config audit only, not runtime audit) - Config audit reflects state at query time; newly created monitors/rules not reflected until next query -``` ## Safety Rules diff --git a/plugins/tsuga/skills/tsuga-audit/references/resource-governance.md b/plugins/tsuga/skills/tsuga-audit/references/resource-governance.md new file mode 100644 index 0000000..c222452 --- /dev/null +++ b/plugins/tsuga/skills/tsuga-audit/references/resource-governance.md @@ -0,0 +1,101 @@ +# Resource Governance Audit + +Covers three administrative Tsuga config resources whose hygiene isn't captured by the telemetry-quality, monitor-coverage, dashboard-hygiene, or routing classes: ingestion API keys, tag policies, and retention policies. Each is individually thin — a handful of fields, a handful of real checks — which is why this reference covers all three instead of splitting into three near-empty files. + +## Example Requests + +- "Are there any unused ingestion keys we should revoke?" +- "Do we have tag policies configured, and are they active?" +- "Which teams or environments have no retention policy?" +- "Audit our resource governance" + +## Required Inputs + +- **Scope** (optional, default: all three resource types, all teams): can be narrowed to a team, or to just one of ingestion keys / tag policies / retention policies. + +## Runtime Docs Lookup + +For docs lookup rationale and docs-error behavior, follow `tsuga-cli`; examples omit `--rationale` for brevity. + +| Need | Fetch | +|---|---| +| Ingestion vs operation keys, ownership/tag model | `tsuga docs get account-and-settings/api-keys` | +| Tag policy product docs | `tsuga docs get account-and-settings/tag-policies` | +| Configuring tag policies | `tsuga docs get account-and-settings/guides/how-to-configure-tag-policies` | +| Retention product docs | `tsuga docs get account-and-settings/retention` | +| Configuring retention policies | `tsuga docs get account-and-settings/guides/how-to-configure-data-retention-policies` | + +## Workflow + +### Ingestion API keys + +1. `tsuga ingestion-api-keys list` — inventory visible key metadata such as name, masked value, owner, tags, and team override fields when present. Do not assume the CLI response exposes usage volume or last-seen fields — "unused" is not independently verifiable via this command unless the installed CLI output clearly includes a usage field. +2. Start from the quality report's `no-unused-ingestion-api-keys` rows (or the ones the orchestrator already fetched) as the primary — effectively only — source of "unused" evidence. Label it `source: quality report (not independently verifiable via CLI)`. +3. As indirect, approximate corroboration only: cross the key's `owner` team against that team's services in `tsuga services list` (`lastSeenAt`, signal counters). This isn't a 1:1 mapping — a key doesn't correspond to exactly one service — so label any finding from this step as indirect, not confirmed. +4. Resolve ownership as usual: `owner` values absent from `tsuga teams list` are stale team references. + +### Tag policies + +5. `tsuga tag-policies list` — fields include `name`, `isActive`, `tagKey`, `allowedTagValues[]`, `isRequired`, `owner`, and `configuration` (`type`: `telemetry` or `tsuga_asset`, plus `assetTypes[]`). `configuration` decides what a policy actually governs — check it before choosing evidence in step 6. Flag inactive policies (`isActive: false`) separately — an inactive policy exists in config but enforces nothing. +6. For an active, required policy, the compliance check depends on `configuration.type`: + - `type: "telemetry"` (`assetTypes` includes `logs`/`metrics`/`traces`): cross `tagKey` against real telemetry — sample recent data with that attribute (`tsuga logs search --query ":*" --max-results 10`, or `tsuga aggregation scalar` grouped by the tag) and check whether observed values fall inside `allowedTagValues`. This is a bounded proxy, not an exhaustive scan — state the sample size and window. + - `type: "tsuga_asset"` with `assetTypes` including `ingestion-api-key`: this governs tags on the ingestion key resource itself, not telemetry — the tag will never appear in log/metric/trace attributes, so a telemetry search is the wrong evidence and will read as a false gap. Cross `allowedTagValues` against the `tags[]` array already pulled from `tsuga ingestion-api-keys list` in step 1, keyed by `tagKey`. + - `type: "tsuga_asset"` with other `assetTypes` (e.g. `rum-public-token`): no CLI resource lists or reads that asset type's tags — label any finding `source: quality report only (no CLI resource for this asset type)` rather than fabricating a check. +7. Resolve ownership as usual, and note `configuration.assetTypes` in any finding so the evidence type is traceable. + +### Retention policies + +8. `tsuga retention-policies list` — fields are `env`, `teamId`, `dataSource`, `durationDays`, `isEnabled`. Cross `tsuga teams list` × known envs × `dataSource` (`logs`/`metrics`/`traces`) to find combinations with no policy at all — these fall back to an org default that may not match compliance or cost intent. Say so rather than assuming the gap itself is wrong; some orgs deliberately rely on the default. +9. Flag `durationDays` values that are outliers relative to peer teams/envs for the same `dataSource` — e.g. one team retaining logs for 3 years next to everyone else's 30 days is worth a question, not an automatic finding. Organizations vary intentionally; state it as "worth investigating." + +## Evidence Rules + +- Every finding cites the command and value that produced it. +- Label ingestion-key "unused" findings as quality-report-only; label telemetry tag-value compliance findings as a sampled proxy, not exhaustive; label asset-tag compliance findings (`configuration.type: "tsuga_asset"`) as cross-referenced against the asset's own `tags[]`, not sampled; label retention-duration outliers as "worth investigating," not a defect. +- Quality-report findings: carry the row's `recommendation` text into Recommended Actions close to verbatim, and prioritize multiple findings by estimated impact — see `tsuga-audit`'s Quality Reports step for the exact rule. + +## Safety Rules + +- Read-only by default. Creating, updating, or deleting any of the three resource types requires the mutation gate: show the proposed change and why, wait for explicit confirmation, apply only after. +- **Revoking an ingestion key is high blast-radius** — it can immediately stop ingestion for every service sending through it. Confirm the key is genuinely unused (not just report-flagged) via the indirect service-signal check before recommending revocation, and call out the blast radius explicitly in the proposed change. +- Never present a tag-value compliance finding as exhaustive; it's a bounded sample. + +## Output Template + +```markdown +## Resource Governance Audit +Scope: / > | As of: | Quality report generated: + +## Summary +Ingestion keys: | Flagged unused (quality report): +Tag policies: | Inactive: | Compliance samples checked: +Retention policies: | Team/env/dataSource combinations with no policy: + +## Ingestion Keys — Findings +| Key | Owner | Note | +|---|---|---| + +## Tag Policies — Findings +| Policy | Tag key | Status | Note | +|---|---|---|---| + +## Retention Policies — Findings +| Team/Env | Data source | Duration | Note | +|---|---|---|---| + +## Findings +| Finding | Evidence | Source | Confidence | +|---|---|---|---| + +## Recommended Actions + +## Limitations +- Ingestion-key usage may not be exposed by the installed CLI; when no usage field is present, "unused" relies entirely on the quality report. +- Tag-value compliance checks are bounded samples, not exhaustive scans. +- Retention-duration comparisons are heuristic (peer outliers), not measured against a stated organizational policy unless one exists. +``` + +## Related Skills / Next Steps + +- `tsuga-cli` — resource command syntax, owner lookups. +- `tsuga-audit` (telemetry quality class) — cross-check tag/attribute compliance findings against broader naming/cardinality issues. diff --git a/plugins/tsuga/skills/tsuga-audit/references/routes.md b/plugins/tsuga/skills/tsuga-audit/references/routes.md new file mode 100644 index 0000000..84ae736 --- /dev/null +++ b/plugins/tsuga/skills/tsuga-audit/references/routes.md @@ -0,0 +1,101 @@ +# Telemetry Routing Audit + +Audits the `routes` resource — team-owned log processing definitions that decide which team owns incoming telemetry and what enrichment runs on it before it's queryable. This is upstream of and distinct from monitor/notification routing: a route decides *who owns this data*; a notification rule decides *who gets alerted*. + +## Example Requests + +- "Audit our log routing" +- "Are there logs that don't match any route?" +- "Which routes are duplicating processing on the same logs?" +- "Does every team have a route for their logs?" +- "Review our route configuration" + +## Required Inputs + +- **Scope** (optional, default: all routes): team, or a specific route ID/name. + +## Runtime Docs Lookup + +For docs lookup rationale and docs-error behavior, follow `tsuga-cli`; examples omit `--rationale` for brevity. + +| Need | Fetch | +|---|---| +| Route model — ownership, ranking, non-exclusive matching | `tsuga docs get process/routes/index` | +| Processor field/parse behavior | `tsuga docs get process/routes/processors` | +| Route API body | `tsuga docs get api/createRoute` and `tsuga docs get api/updateRoute` | +| Missing/changed log troubleshooting | `tsuga docs get process/guides/how-to-investigate-a-missing-or-changed-log` | +| TQL syntax used in Source query | `tsuga docs get explore/query-syntax` | + +## Workflow + +1. Resolve scope. `tsuga routes list` has no server-side filter — pull the full list and filter locally by `owner`/`tags`. + +2. Start from the quality report (same convention as the other references): filter rows to `ruleId` in `team-has-route`, `unrouted-logs`, `no-over-routed-logs` for the requested team, or use the rows the orchestrator already fetched. + +3. **Understand the matching model before flagging anything.** Routes are not exclusive: Tsuga checks every *enabled* route top to bottom by `rank`, and every route whose `query` matches a log runs on it — later routes see and can modify fields written by earlier ones. This is intentional layered enrichment, not a bug. Don't flag every log matched by more than one route as a problem; use the quality report's `no-over-routed-logs` finding and its `recommendation` text to identify *excessive* overlap, not any overlap. + +4. `tsuga routes list` — for each route, note `owner`, enabled/disabled state, `rank`, and `query`. Cross `owner` against `tsuga teams list`: an owner ID absent from `teams list` is a stale team reference, same convention used for monitors and dashboards. + +5. Team routing coverage: cross `tsuga teams list` against distinct enabled-route `owner` values. A team with zero enabled routes is a coverage gap — its telemetry either falls through unrouted or depends entirely on another team's route matching it incidentally. + +6. Unrouted-logs check: sample recent logs (`tsuga logs search --max-results 10`) and compare against the union of enabled route `query` strings for an intuition check, but treat the quality report's `unrouted-logs` finding as the primary evidence — it's computed over the full ingest window, not a 10-row sample. + +7. Disabled routes: list them separately. New routes start inactive by design until reviewed, so a route existing but disabled is a different finding than no route existing at all — don't conflate "unactivated" with "missing." + +## Evidence Rules + +- Every finding cites the command and value that produced it. +- "Over-routed" requires the quality report's `no-over-routed-logs` finding, not a manual count of overlapping `query` strings — multiple matches are expected by design; only the report's threshold identifies excess. +- Route config changes propagate only to logs ingested after the cluster picks up the change — a route audit reflects current config, not necessarily what produced already-ingested historical logs. State this when a finding depends on route config explaining past log behavior. +- Quality-report findings: carry the row's `recommendation` text into Recommended Actions close to verbatim, and prioritize multiple findings by estimated impact — see `tsuga-audit`'s Quality Reports step for the exact rule. + +## Safety Rules + +- Read-only by default. Creating, updating, enabling/disabling, re-ranking, or deleting a route requires the mutation gate: show the proposed change and why, wait for explicit confirmation, apply only after. +- Changing a route's Source query requires Global Admin or an admin of the route's owning team in the product — if a proposed fix needs this, say so rather than assuming the acting credential has permission. +- Never claim a route "fixed" missing data without confirming with a fresh `logs search` after the cluster picks up the change — propagation isn't instant. + +## Output Template + +```markdown +## Telemetry Routing Audit +Scope: > | As of: | Quality report generated: + +## Summary +Routes audited: | Enabled: | Disabled: +Teams with zero enabled routes: + +## Unrouted Logs (quality report evidence) + + +## Over-Routed Logs (quality report evidence) + + +## Stale Team References +| Route | Owner (missing team ID) | +|---|---| + +## Teams Without Any Enabled Route +| Team | Note | +|---|---| + +## Disabled Routes +| Route | Owner | Note | +|---|---|---| + +## Findings +| Finding | Evidence | Source | Confidence | +|---|---|---|---| + +## Recommended Actions + +## Limitations +- Route matching is non-exclusive by design; overlap alone isn't a defect — only the quality report's threshold identifies excess. +- Route changes apply prospectively; this audit doesn't explain historical log behavior from before the current config was saved. +- `routes list` has no server-side filter; large route inventories are filtered client-side. +``` + +## Related Skills / Next Steps + +- `tsuga-cli` — resource command syntax, TQL for Source query, owner lookups. +- `tsuga-audit` (monitor coverage class) — once telemetry is routed to a team, check whether it's also monitored and alerted on. diff --git a/plugins/tsuga/skills/tsuga-audit-telemetry-quality/SKILL.md b/plugins/tsuga/skills/tsuga-audit/references/telemetry-quality.md similarity index 90% rename from plugins/tsuga/skills/tsuga-audit-telemetry-quality/SKILL.md rename to plugins/tsuga/skills/tsuga-audit/references/telemetry-quality.md index 392dbd2..dc6e2d6 100644 --- a/plugins/tsuga/skills/tsuga-audit-telemetry-quality/SKILL.md +++ b/plugins/tsuga/skills/tsuga-audit/references/telemetry-quality.md @@ -1,11 +1,6 @@ ---- -name: tsuga-audit-telemetry-quality -description: "Use when auditing telemetry quality for logs, metrics, traces, or resource identity: log structure/correlation, metric naming/units/temporality/instrument type/cardinality, span naming/status/kind/links/noisy spans, semantic naming/cardinality, telemetry quality reviews, quality reports, downstream metric usage, noisy signals, missing labels, malformed attributes, resource drift, source labels, broken correlation, or recommendations to rename metrics or right-size high-cardinality attributes." ---- +# Telemetry Quality Audit -# Tsuga Audit Telemetry Quality - -Use this as the read-only quality audit workflow for telemetry Tsuga is receiving. Keep SKILL.md focused on evidence, classification, and safety; use runtime docs for product, OTel, and language details. +Use this as the read-only quality audit workflow for telemetry Tsuga is receiving. Keep this reference focused on evidence, classification, and safety; use runtime docs for product, OTel, and language details. ## Required Inputs @@ -61,6 +56,7 @@ Use `otel-instrumentation` for confirmed-language implementation patterns after - Cardinality group-by results are proxies, not exact measurements; state query limits. - Source-code findings cite file path and line. If not confirmed in Tsuga, label as `Recommendation (not verified in Tsuga)`. - Root cause requires at least two corroborating signals. A single signal is only consistent with a hypothesis. +- Quality-report findings: carry the row's `recommendation` text into Recommended Actions close to verbatim, and prioritize multiple findings by estimated impact — see `tsuga-audit`'s Quality Reports step for the exact rule. ## Safety diff --git a/plugins/tsuga/skills/tsuga-build-dashboard/SKILL.md b/plugins/tsuga/skills/tsuga-build-dashboard/SKILL.md index 95eb5dd..74cf738 100644 --- a/plugins/tsuga/skills/tsuga-build-dashboard/SKILL.md +++ b/plugins/tsuga/skills/tsuga-build-dashboard/SKILL.md @@ -158,6 +158,6 @@ Time preset: ## Related Skills / Next Steps - `tsuga-cli` — filter syntax, counter math, aggregation body construction, metric discovery, dashboard CRUD, time formats -- `tsuga-audit-telemetry-quality` — audit metric and telemetry quality before dashboarding +- `tsuga-audit` — audit metric and telemetry quality before dashboarding - `tsuga-investigate-service-health` — the health triage workflow that dashboards should operationalize - `tsuga-debug-telemetry-ingestion` — if expected metrics or signals are missing diff --git a/plugins/tsuga/skills/tsuga-cli/SKILL.md b/plugins/tsuga/skills/tsuga-cli/SKILL.md index 3e74e28..1b3d189 100644 --- a/plugins/tsuga/skills/tsuga-cli/SKILL.md +++ b/plugins/tsuga/skills/tsuga-cli/SKILL.md @@ -21,6 +21,7 @@ Use CLI help and `--generate-skeleton` first for CLI CRUD payload shape. Fetch d | Traces product/query docs | `tsuga docs get explore/traces` | | Monitors product docs | `tsuga docs get alert/monitors/index` | | Dashboards product docs | `tsuga docs get visualize/dashboards/index` | +| Metric aggregation choice / temporality edge cases | `tsuga docs get visualize/guides/how-to-choose-a-metric-aggregation` | | Aggregation API bodies | `tsuga docs get api/aggregateScalar` and `tsuga docs get api/aggregateTimeseries` | | Logs API body | `tsuga docs get api/searchLogs` | | Traces API body | `tsuga docs get api/searchSpans` | @@ -126,6 +127,8 @@ Run `tsuga metrics get ` before choosing aggregate/function. Wrong math pr | Counter, cumulative | `sum` | `rate` or `increase` | | Histogram | `percentile` with `field` and `percentile` | none | +This table is a quick reference, not the full model. For temporality nuances — why cumulative counters need an upstream `cumulativetodelta` processor, and how Tsuga sums a gauge across dimensions within a bucket (average within each dimension, then sum — not a naive sum, and not Prometheus's instant-vector `sum`) — fetch `tsuga docs get visualize/guides/how-to-choose-a-metric-aggregation` rather than hand-deriving it. + When the right metric is unclear, inspect existing dashboards before the metric catalog; dashboards contain validated metric/filter/aggregation combinations. ## Resource And API Operations diff --git a/plugins/tsuga/skills/tsuga-cli/references/playbooks/find-owner-and-context.md b/plugins/tsuga/skills/tsuga-cli/references/playbooks/find-owner-and-context.md index 53e8276..1649bdf 100644 --- a/plugins/tsuga/skills/tsuga-cli/references/playbooks/find-owner-and-context.md +++ b/plugins/tsuga/skills/tsuga-cli/references/playbooks/find-owner-and-context.md @@ -63,5 +63,5 @@ Env: | Last seen: | Sources: ## Related / Next Steps - `tsuga-investigate-service-health` skill — health check for the identified service -- `tsuga-audit-monitor-coverage` skill — check alerting coverage for the team's services +- `tsuga-audit` skill — check alerting coverage for the team's services - `otel-instrumentation` skill — full observability review for the service via runtime docs diff --git a/plugins/tsuga/skills/tsuga-cli/references/playbooks/reliability-review.md b/plugins/tsuga/skills/tsuga-cli/references/playbooks/reliability-review.md index b7a55f8..be2f046 100644 --- a/plugins/tsuga/skills/tsuga-cli/references/playbooks/reliability-review.md +++ b/plugins/tsuga/skills/tsuga-cli/references/playbooks/reliability-review.md @@ -66,7 +66,6 @@ Teams flagged in both this review and a monitor coverage audit: ## Related / Next Steps -- `tsuga-audit-monitor-coverage` skill — alerting gap audit +- `tsuga-audit` skill — alerting gap audit, or metric design and telemetry quality audit - `otel-instrumentation` skill — full observability audit for a specific service via runtime docs -- `tsuga-audit-telemetry-quality` skill — metric design and telemetry quality audit - `tsuga-cli` — identify team owners for failing services; use the bundled owner/context playbook when needed diff --git a/plugins/tsuga/skills/tsuga-debug-telemetry-ingestion/SKILL.md b/plugins/tsuga/skills/tsuga-debug-telemetry-ingestion/SKILL.md index e941704..71f7ca4 100644 --- a/plugins/tsuga/skills/tsuga-debug-telemetry-ingestion/SKILL.md +++ b/plugins/tsuga/skills/tsuga-debug-telemetry-ingestion/SKILL.md @@ -97,5 +97,5 @@ Use `otel-instrumentation` for confirmed-language source fixes after the ingesti - `otel-instrumentation` - language-specific SDK setup, exporter config, log correlation, and source fixes. - `otel-collector` - Collector pipeline, exporter, processor, OTTL, and redaction debugging. -- `tsuga-audit-telemetry-quality` - audit signal shape after telemetry is arriving. +- `tsuga-audit` - audit signal shape after telemetry is arriving. - `signal-choice-advisor` - redesign signal choice, semantic naming, or cardinality after the ingestion issue is isolated. diff --git a/plugins/tsuga/skills/tsuga-investigate-errors/SKILL.md b/plugins/tsuga/skills/tsuga-investigate-errors/SKILL.md index 8df4380..d3f3b31 100644 --- a/plugins/tsuga/skills/tsuga-investigate-errors/SKILL.md +++ b/plugins/tsuga/skills/tsuga-investigate-errors/SKILL.md @@ -108,4 +108,4 @@ Source: aggregation scalar, filter: context.service.name:"" level:ERROR