From 80fa43c1aca89e9b3ee60a9269049aa1840e32e4 Mon Sep 17 00:00:00 2001 From: agenticode <16611333+agenticode@users.noreply.github.com> Date: Wed, 26 Aug 2026 15:28:45 +0900 Subject: [PATCH] cmd, pkg/domain: wire rds, backtest and explain into the binary MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The three packages merged earlier today all deferred their wiring, so a user building today could not reach any of them. This closes that gap: pkg/domain learns the rds kind, and cmd/kilter gains rds, backtest and explain subcommands. The backtest live path refuses honestly, because pkg/store keeps only the latest snapshot and has no history to replay. Closes a real hole found on the way: the account-wide commitment baseline accepted domain-supplied usage lines unchecked, so an empty ID double-counted, a duplicate ID overwrote another domain's line, and a zero rate priced real usage at nothing — all three overstate absorption and therefore savings. Invalid lines are now dropped with a warning, and a control test proves this is a no-op for every shipped domain. Also fixes a renderer that printed the oracle gap at 100x because Scorecard.OracleGapPct is already scaled while its doc comment says otherwise. The two test-fixture updates are forced, not loosened: five tests used the literal "rds" as their stand-in for an unknown kind, which it no longer is, and one asserted the number of wired domains. Every assertion is unchanged. Co-authored-by: kording <74226694+kording@users.noreply.github.com> --- cmd/WIRING-FINDINGS.md | 528 +++++++++++++++++++++++ cmd/kilter/backtest.go | 527 +++++++++++++++++++++++ cmd/kilter/backtest_test.go | 291 +++++++++++++ cmd/kilter/domains.go | 165 +++++++- cmd/kilter/domains_test.go | 4 +- cmd/kilter/explain.go | 609 +++++++++++++++++++++++++++ cmd/kilter/explain_test.go | 479 +++++++++++++++++++++ cmd/kilter/main.go | 16 +- cmd/kilter/rds.go | 163 +++++++ cmd/kilter/rds_test.go | 553 ++++++++++++++++++++++++ cmd/kilter/rdsfixture_test.go | 225 ++++++++++ cmd/kilter/testdata/rds-account.json | 1 + pkg/domain/composite_test.go | 4 +- pkg/domain/domain.go | 8 +- pkg/domain/domain_test.go | 2 +- pkg/domain/hostilerds_test.go | 240 +++++++++++ pkg/domain/rds/rds.go | 120 ++++++ pkg/domain/rds/rds_test.go | 170 ++++++++ pkg/domain/registry_test.go | 2 +- pkg/domain/report_test.go | 2 +- 20 files changed, 4092 insertions(+), 17 deletions(-) create mode 100644 cmd/WIRING-FINDINGS.md create mode 100644 cmd/kilter/backtest.go create mode 100644 cmd/kilter/backtest_test.go create mode 100644 cmd/kilter/explain.go create mode 100644 cmd/kilter/explain_test.go create mode 100644 cmd/kilter/rds.go create mode 100644 cmd/kilter/rds_test.go create mode 100644 cmd/kilter/rdsfixture_test.go create mode 100644 cmd/kilter/testdata/rds-account.json create mode 100644 pkg/domain/hostilerds_test.go create mode 100644 pkg/domain/rds/rds.go create mode 100644 pkg/domain/rds/rds_test.go diff --git a/cmd/WIRING-FINDINGS.md b/cmd/WIRING-FINDINGS.md new file mode 100644 index 0000000..d3a54ee --- /dev/null +++ b/cmd/WIRING-FINDINGS.md @@ -0,0 +1,528 @@ +# P2 — wiring: three shipped packages become reachable from the binary + +`pkg/rds`, `pkg/backtest` and `pkg/explain` shipped last cycle and **none of +them was reachable from `kilter`**. Each deferred its wiring to a FINDINGS.md, +exactly as PR#29 (j1-wire) had to clean up once before. A user who built the +binary could not run any of them. + +Four commands now do: + +``` +kilter domains report --rds-fixture … observe → refuse → report, across the RDS domain +kilter backtest --demo score a policy against history; Gate; CI exit code +kilter explain --workload … --container … why the engine would resize this container +kilter why-cost --from … --to … an additive, individually-citable cost decomposition +``` + +**Status:** `gofmt -l ./cmd ./pkg/domain` empty, `go vet ./...`, `go build ./...`, +`go test -race -count=1 ./cmd/... ./pkg/domain/...` and `go test -race -short ./...` +(35 packages) all green. **`go.mod` and `go.sum` are unchanged** — every import +added is stdlib or intra-repo. + +| | | +|---|---| +| New production code | 1,419 lines across `cmd/` and `pkg/domain/rds/` | +| New tests | 1,958 lines, 40 test functions | +| Coverage | `pkg/domain` **89.3 %**, `pkg/domain/rds` **89.5 %**, `cmd/kilter` 45.1 % | + +``` +pkg/domain/domain.go + the RDS kind (the two-line core change §6.1 owed) +pkg/domain/rds/ the pkg/rds adapter: registration + the usage-line projection +pkg/domain/hostilerds_test.go the adversarial suite: a domain that lies through four seams +cmd/kilter/rds.go the three read seams over a recorded account + the rate override +cmd/kilter/backtest.go kilter backtest: traces wired, live history refused by name +cmd/kilter/explain.go kilter explain + kilter why-cost, both calling Verify before serving +cmd/kilter/domains.go + rds registration, the ledger splice, and its output check +``` + +--- + +## 1. What is reachable now + +### 1.1 RDS — a domain whose entire product is refusals + +``` +$ kilter domains report --now 2026-08-26T12:00:00Z --domain rds \ + --scope 000000000000/us-east-1 --region us-east-1 \ + --rds-fixture cmd/kilter/testdata/rds-account.json + + DOMAIN STATE ACTUATION TARGETS RECS REFUSED CLAIMABLE GROSS + rds ready report-only 7 0 27 $0.00 $0.00 + + What kilter declined to do, and why + allocated-storage-cannot-shrink 4 refused + instance-class-change-is-a-failover 4 refused + no-storage-performance-model 4 refused + unverified-rate 4 refused + freeable-memory-is-page-cache 2 refused + aurora-not-supported 1 refused + … +``` + +Zero recommendations and zero claimable dollars is the **deliverable**, not a +gap. `pkg/rds/FINDINGS.md` §2: the instance class is where the money is and +changing it is a failover; allocated storage cannot shrink; `FreeableMemory` is +`MemAvailable`. The domain proposes nothing, so its whole output arrives +through the `domain.Refuser` seam — the seam j1-wire added for exactly this +case, now carrying its first domain that has nothing else to say. + +`--rds-detail` additionally prints `pkg/rds`'s own report, whose layout is the +opposite of the aggregate's on purpose: refusals first, money second, because +in this domain the refusal *is* the finding and a reader who sees a dollar +figure first reads the refusal as a caveat on a recommendation that does not +exist. + +### 1.2 `kilter backtest` — the falsifiability harness + +``` +$ kilter backtest --demo regime-change --workloads 3 --compare enforced.json + + policy 69955a2cdd2004fd + scored 21 decisions 18 refusals 3 (good 0, idle 3) + safety memViolations 3 cpuStarvation 3 oomKills 0 + efficiency oracleGap 94.5% (applied 5.7%) claimed/realized 0.96 + regret $303.86 (resource $3.86 + risk $300.00) + + candidate 69955a2cdd2004fd + scored 21 decisions 15 refusals 6 (good 0, idle 6) + safety memViolations 3 cpuStarvation 0 oomKills 0 + efficiency oracleGap 129.6% (applied 16.1%) claimed/realized 1.00 + regret $157.21 (resource $7.21 + risk $150.00) + + Gate + ACCEPTED: the candidate dominates on the terms §4.6 defines +``` + +That is `pkg/backtest/FINDINGS.md`'s headline A/B, reproduced field for field +from the binary: wiring `pkg/decision` into the recommendation path removes +**all** CPU starvation and halves regret, bought with $3.35 of extra idle +headroom. `TestWiringTheDecisionLayerIsAnImprovementThroughTheCLI` asserts the +win *and* that it was paid for — a scorecard reporting only the win would be an +advertisement. The four shipped goldens (steady $2.74, diurnal $4.02, bursty +$2.90, regime-change $202.57) are pinned by +`TestBacktestReproducesTheShippedGoldens`. + +### 1.3 `kilter explain` and `kilter why-cost` + +``` +$ kilter explain --kube-snapshot cluster.json \ + --workload Deployment/default/api --container api + +kilter explain — Deployment/default/api/api over [2026-08-23T12:15:00Z, 2026-08-26T12:00:01Z] + +Sizing: api request 1000m/8192Mi → 324m/7802Mi, limit 0m/0Mi → 0m/0Mi. +Because: + behavior-class behavior class "steady" selects the sizing policy applied [dig/1/…@1787486400000000000] + usage-history 287 samples over 71h45m0s: cpu p95 300m / max 300m, memory p95 6.5GiB / max 6.5GiB [dig/1/…] +``` + +Both commands **call `Verify` before serving**, and treat a failure as an error +rather than rendering an answer with a dangling citation. +`pkg/explain/FINDINGS.md` §4 is explicit that nothing else enforces §5.7's +"citations must resolve to ids the session actually fetched"; `Verify` exists +and someone has to call it. That someone is now `runWhyCostTo` and +`runExplainTo`. + +--- + +## 2. The core change, and the two test fixtures it made stale + +`pkg/domain/domain.go` grew the row `pkg/rds/FINDINGS.md` §6.1 specified: + +```go +RDS Kind = "rds" +var kinds = []Kind{EC2, ECSFargate, K8sFargate, K8sNodes, Lambda, RDS} +``` + +**§6.1's claim was checked, not trusted.** `TestKindIsHonestAboutRegistration` +does pass both before and after: it asserts the property that holds either way +(registered ⇒ the core still refuses to plan steps; not registered ⇒ `Register` +fails with `unknown kind` and the standalone `Report` path works). Verified by +running it on both sides of the edit. + +What §6.1 did **not** anticipate is that five other tests used the literal +string `"rds"` as their stand-in for *a kind outside the closed set*. Those +placeholders went stale the moment the kind became real. Every assertion is +unchanged; only the sample string moved, to `"quantum-annealer"`: + +| File | What it asserts (unchanged) | +|---|---| +| `pkg/domain/domain_test.go:26` | unknown kinds do not validate | +| `pkg/domain/composite_test.go:159,162` | a composite cannot be built for an unknown kind | +| `pkg/domain/registry_test.go:112` | `Register` rejects an unknown kind | +| `pkg/domain/report_test.go:238` | `Report.Validate` rejects an unknown kind | +| `cmd/kilter/domains_test.go:467` | `--domain` rejects an unknown domain | + +One further count moved: `TestNoDomainCanPlanAStepInThisBuild` asserted +`len(plans) == 4`, "one per wired domain". There are five wired domains now, so +it reads `5`. The loop body — every plan refuses, none is actuatable, none has +steps — is untouched, and it is where the test's meaning lives. No expectation +was weakened; a count of wired domains cannot survive wiring a domain. + +--- + +## 3. Report-only enforcement: what the core guarantees for `rds` + +`pkg/rds` is structurally read-only — no mutating API appears anywhere in the +package, `Health` is unconditionally report-only, `PlanSteps` refuses +unconditionally. **None of that is why an RDS step cannot run**, and the +distinction matters now that `rds` is a public registry key anything in-process +can claim. + +Three walls, in the order a step meets them, none of which a domain can write to: + +1. **`Registry.PlanSteps` checks `Health` before the domain is consulted.** An + honest report-only domain's `PlanSteps` is never called at all + (`TestAnHonestRDSDomainIsAlsoRefused`). +2. **`validateSteps` checks the output.** A domain registered as `rds` that + hands back a step labelled `ec2` — where a real actuator *is* wired — is + stopped with `ErrWrongDomain`, and the victim's actuator is never reached. +3. **`Registry.Execute` routes through the actuator table.** There is no `rds` + row and there cannot be one. `Execute` and `Revert` both return + `ErrReportOnly`. + +`TestAHostileRDSDomainCannotActuateOrBorrowAnotherDomainsActuator` builds a +domain that lies about all three and asserts each wall independently, including +that the test actually reached the domain's `PlanSteps` (an adversarial test +that never reaches the attack proves nothing). + +Two seams the RDS wiring newly *depends* on were attacked as well, because both +are new attack surface: + +- **`Refuser`.** This is the first domain whose entire output is refusals. A + domain returning refusals pointed at another domain's targets would put words + in that domain's mouth — a `not-gp2` line filed under `ec2` that `pkg/ebs` + never said, in a report a human is asked to act on. The registry re-stamps + the producing kind, and `TestAHostileDomainCannotAttributeFindingsToAnotherDomain` + asserts the victim's row counts none of them. +- **`SetSavings`, bypassed.** A domain can assign `Net`/`Gross` directly and + skip the clamp that makes `Net ≤ Gross` mechanical. Nothing in `Summarize` + re-checks it, by design — `Report.Validate` is the gate, and the CLI calls it + before printing. `TestAFabricatedSavingIsCaughtByReportValidate` pins that. + +### 3.1 A real hole this unit found and closed: the usage-line seam + +`cmd/`'s account-wide commitment baseline is built from `UsageLines(now, +ledger)` — a **domain-supplied output that lands in a structure every other +domain's net savings are computed against**. Until this unit nothing checked +it. That is the same shape as the hole j1-wire closed one level down: inputs +filtered, outputs trusted. + +Three ways it goes wrong, all silent, all in the *overstating* direction: + +- an **empty ID** never matches in `domain.Ledger`'s splice and is always + appended, so two anonymous lines for one resource double-count the usage + available to absorb a commitment; +- a **duplicate ID** from a second domain *replaces* the first domain's line + rather than adding to it, so one domain silently rewrites another's + contribution; +- a **non-positive or non-finite rate or quantity** prices real usage at + nothing, which again makes a commitment look more absorbed than it is. + +More apparent absorption means a larger claimed saving. `badBaselineLine` +(`cmd/kilter/domains.go`) drops such a line and records a collection warning — +dropping is the conservative direction, because less baseline usage means more +apparent stranding and a *smaller* claim. +`TestPoisonedUsageLinesNeverReachTheAccountWideBaseline` builds all five +variants; `TestTheShippedDomainsContributeCleanBaselineLines` is the control +that the gate is a no-op for every domain actually wired, so it is not silently +shrinking a real baseline. + +--- + +## 4. Every adapter written, and the mismatch that forced it + +### 4.1 `rdsFixtureFile` — because `rds.Fixture` has fields JSON cannot carry + +`rds.Fixture` implements all three read seams and is exported precisely so +`cmd/` can test against a recorded account. It cannot be decoded from JSON +directly: it carries four `error` fields (`InstancesErr`, `ClustersErr`, +`TagsErr`, `MetricsErr`, `ReservationsErr`) and a `Calls` counter struct, none +of which have a JSON representation. Decoding straight into it would produce a +file format with five fields that silently cannot be set and one that pretends +to be input. + +`rdsFixtureFile` carries the data half with explicit json tags, and turns the +two seam-absence cases into the booleans the §6.2 IAM table actually describes: +`noMetricsAPI` (no `cloudwatch:GetMetricData` ⇒ every instance refuses with +`no-metric-evidence`, which is a complete report) and `noCommitmentAPI` (no +`rds:DescribeReservedDBInstances` ⇒ net == gross, which under-claims). + +### 4.2 `pkg/domain/rds.Domain.UsageLines` — because a domain must not build an account-wide view + +`pkg/rds` ships `UsageLines` as a pure function over one instance and stops +there on purpose; `pkg/domain/ledger.go`'s argument is that a domain knows only +its own targets and must not be tempted to construct the account-wide picture. +The adapter projects a whole `Report` into baseline lines, contributing only +the assessments `pkg/rds` could **price**. An excluded instance (Aurora, a +cluster member, an unknown engine, `mode=off`) never reaches the price step, so +`CostKnown` is false and it contributes nothing — which is right, because a +zero-rate baseline line makes a Reserved DB Instance look like it is absorbing +usage that costs nothing. + +### 4.3 `policyFile` — because the three policy configs have no json tags + +None of `recommend.Config`, `plan.Config` or `decision.Config` carries json +tags, so decoding into them gives a file whose keys are Go field names and +whose durations are integer nanoseconds. Worse, **omitted fields decode as +zero**: a policy file where leaving out `cpuHeadroom` silently means "headroom +1.0" is a footgun with a business consequence. Every field in `policyFile` is +therefore a **pointer** overlaid onto the package defaults, so absent means +default and present means present. Unknown fields are rejected (a knob +misspelled and silently ignored produces a scorecard for a policy nobody ran), +and a bare number where a duration belongs is rejected by name because Go would +read it as nanoseconds and nobody ever meant that. + +`enforceDecisionRefusals` is modelled as part of the **policy**, not the +scoring knobs, because it is pending production wiring rather than a yardstick. +That is what turns "should we wire the decision layer in?" into an A/B through +`Gate` instead of an opinion. + +### 4.4 `basisFrom` — and the three things §2 said the wiring must get right + +All three are implemented and each has a note in the code saying why: + +1. **Fargate nodes are excluded** from `Groups`. A Fargate "node" is a + single-pod VM billed per quantized pod; pricing it per node shape would + inflate the fleet total and put the error *inside a term*. Excluding it moves + the gap to the residual, where it is visible. + `TestFargateIsExcludedFromTheCompositionAndLandsInTheResidual` asserts the + residual is non-zero, every priced term is zero, and a note is attached. +2. **An empty fleet is `&CostBasis{At: t}`, not `nil`.** `nil` is returned only + for a snapshot that does not exist. +3. **Namespace demand is *requested* capacity.** `clampedPodRequests` mirrors + `pkg/plan`'s unexported `clampedRequests` exactly (negatives clamped to + zero), duplicated rather than re-derived because §2 is explicit that a third + definition of "requested capacity" is the thing to avoid, and `pkg/plan` is + not this unit's to change. + +### 4.5 `loadLedgerActions` — §1's projection, field for field + +Both things §1 insists on are enforced and tested. `Applied` is exact: only +`Mode == "apply"` with `Done > 0`, and only `actuate.StatusDone` steps are +counted — `StatusDryRun` is deliberately excluded even though `actuate.Report` +counts it as done for its own purposes, because a preview moved no money and +attributing a cost change to it is the classic attribution lie with a plan +attached. `NodesAdded` stays 0 because no plan type provisions a node today. +`Finished` has no ledger field, so it is left zero and `pkg/explain` falls back +to `At`. + +--- + +## 5. Bugs found while wiring + +### 5.1 `Scorecard.OracleGapPct` is already scaled — the doc comment says otherwise + +`scorecard.go:371` computes `round6(meanSorted(gapsAll) * 100)`, while the field's +doc comment states the unscaled ratio `(cost(outcome) − cost(oracle)) / cost(oracle)`. +The first draft of the renderer multiplied by 100 again and printed a +**9,445 % oracle gap** for a trace whose real gap is 94.5 %. Caught by +comparing the rendered output against `pkg/backtest/FINDINGS.md`'s published +goldens, which is the reason those goldens are worth publishing. +`TestOracleGapIsRenderedInTheUnitsTheScorecardUses` is the regression. + +**Reported, not fixed** — the fix belongs in `pkg/backtest`, and it is a doc +comment, not the arithmetic: the field name says `Pct` and the value is a +percentage. `Tolerance.MaxOracleGapIncreasePct`'s "2 pts" default confirms the +percentage reading. + +### 5.2 The half-open window has to be honoured on both sides of the seam + +`explain.WhyCost` filters its timeline to `[From, To)`, matching +`pkg/evidence`'s convention. The first draft observed timeline points over the +**closed** interval, so a snapshot at exactly `--to` was recorded and then +ignored — the CLI's own "needs two observations" check passed while `WhyCost` +failed with a confusing message about one timeline point. Both sides now use +`[from, to)`, the usage text says `--to` is exclusive, and the error message +says so too. + +A related correctness fix: the two `CostBasis` edges are the **first and last +snapshots that produced a timeline point in the window**, not the newest +snapshot on either side of the requested boundary. A composition describing +`t+24h` against a measurement taken at `t+12h` would push a real, explainable +fleet change into the residual and call it unexplained. + +--- + +## 6. What is still NOT reachable, and why + +### 6.1 The live RDS collector — blocked on `go.mod` + +`pkg/rds/FINDINGS.md` §6.2 owes `cmd/` an SDK adapter over `*rds.Client` and +`*cloudwatch.Client`. It **cannot be written in this build**: +`github.com/aws/aws-sdk-go-v2/service/rds` and `.../service/cloudwatch` are not +in `go.mod`, and adding them is a `go.mod`/`go.sum` change this unit may not +make. (`go.mod` has `service/ec2`, `service/pricing` and `service/autoscaling` +only.) + +What replaces it is not a stub. `--rds-fixture` drives the **real collector**: +`rds.NewCollector` over the recorded account, `Collect`, the real window clamp, +the real `GetMetricData` batching and ID routing, real pagination (the shipped +fixture uses `pageSize: 3`, so it spans three pages), real truncation. Every +line of `pkg/rds/collect.go` a live credential would exercise is exercised +here. What is missing is the field copy between an SDK struct and a struct with +the same field names — which §6.2 describes as "a field copy with no +interpretation in it". + +**Next job:** add the two modules, then write ~200 lines of mechanical +adapter per seam. The IAM actions are `rds:DescribeDBInstances`, +`rds:DescribeDBClusters`, `rds:ListTagsForResource` (required — the +`kilter.dev/mode` guardrail is unreachable without it), +`cloudwatch:GetMetricData` and `rds:DescribeReservedDBInstances` (both +optional, both with a documented degraded report). Two conversions are already +done inside `pkg/rds` and **must not be redone**: reservation amortization +(`reservationFromRecord`) and deployment topology (`DBInstance.Deployment`). + +### 6.2 `kilter backtest --cluster` — blocked on snapshot-history persistence + +Refused by name, with the seam spelled out and a non-zero exit: + +``` +$ kilter backtest --cluster prod +kilter backtest: backtest --cluster prod: refused — snapshot history is not persisted. + +pkg/store keeps only the LATEST snapshot per cluster (SaveSnapshot/LoadSnapshot +are keyed by cluster, not by time) … +What it needs: a time-keyed snapshot bucket in pkg/store — SaveSnapshotAt(snap) +and Snapshots(cluster, from, to) — bounded the way the rest of the substrate is, +plus an adapter implementing backtest.SnapshotSource. +``` + +**`pkg/store` was deliberately left untouched.** The additive interface was in +scope, and it was still the wrong thing to add: nothing writes to a time-keyed +bucket today (`pkg/api`'s ingest path is where a write would have to go, and +`pkg/api` is not this unit's), so the bucket would be dead code and +`backtest --cluster` would go from an honest refusal to a scorecard over an +empty history. The refusal is the better artefact until the write side lands. + +The failure mode this refusal exists to prevent is worth stating plainly: +scoring the one snapshot that *does* exist would produce a `Scorecard` with the +same shape, the same field names and the same confident tone as a real one. +`regret $0.00` reads as "the policy is perfect", not as "nothing was replayed". +`TestBacktestLiveHistoryRefusesRatherThanScoringOneSnapshot` asserts no +scorecard is printed. + +Storage note for whoever builds it, from §"Seams this unit needed": at a +5-minute cadence a 30-day window is 8,640 snapshots per cluster. Full snapshots +are far too large; the realistic shape is keyframe-plus-delta, or a reduced +replay snapshot carrying only what `recommend` and `plan` read. + +### 6.3 The explain/why-cost API routes — blocked on `pkg/api` + +`pkg/explain/FINDINGS.md` §4 asks for: + +``` +GET /api/v1/clusters/{id}/why-cost?from=&to= → *explain.Attribution +GET /api/v1/clusters/{id}/explain?subject= → *explain.Explanation +``` + +**Not built.** `pkg/api` is outside this unit's scope, and the routes cannot be +added from `cmd/` by wrapping `Brain.Handler()` either — which is worth +recording, because that wrap *looks* available and is not: + +- **`api.Brain` holds no evidence substrate.** `grep 'evidence\.' pkg/api/brain.go` + returns nothing. Both entry points need one — `WhyCost` needs + `Store.Timeline`, `BuildExplain` reads the substrate through the dossier + builder — and, critically, `Verify` needs *the same store that produced the + answer* to re-serve every citation. A route served from a store the brain + never populated would 500 on every request, which is worse than no route. +- **Brain's internals are unexported.** `Handler()`, `Ledger()`, `Plan()`, + `Recommendations()` and `Clusters()` are the whole surface; the recommender, + the last snapshot and the pricing catalog are not reachable. + +**Next job**, in order: give `api.Brain` an `*evidence.Memory` and populate it +from `Ingest` (samples, a timeline point per snapshot, and the deploy/OOM +events the collectors already carry); persist it through +`evidence.Memory.MarshalCheckpoint`/`FromCheckpoint` into `pkg/store`; then add +the two routes, projecting `LedgerEntry` with the mapping in +`cmd/kilter/explain.go`'s `loadLedgerActions` (which is that projection, written +and tested — lift it rather than re-derive it) and calling `Verify` before +serving. That evidence store is the same prerequisite §6.2 needs for its write +side, so the two jobs share most of their work. + +### 6.4 Smaller gaps + +- **`kilter explain` fills `Rec` and leaves `Verdict` nil.** `pkg/recommend` + does not import `pkg/decision` (`pkg/backtest`'s finding, still true), so + there is no verdict to read out of the production path. The payload's + `Action` is therefore `unknown` and `Refusal` is nil. Filling them means + either the `Recommender.Verdicts(snap)` seam `pkg/backtest` asked for, or + evaluating `decision.Evaluate` at a call site production does not have — + which would be a *different* answer from the one production gives, so it is + deliberately not done. +- **`observeUsage` leaves throttle, restart and OOM at zero.** `model.Usage` + carries CPU, memory and a window and nothing else. Those signals reach the + substrate as evidence events from other collectors; inventing a zero throttle + ratio would be a claim nobody measured, and "signal absent" is a state + `pkg/decision` already knows how to read. +- **`why-cost` prices Fargate into the residual.** `pkg/pricing.SnapshotCost` + includes Fargate pods; the composition excludes Fargate nodes, so the + difference is reported as residual with a note. Correct and coarse; the + natural extension is §"Deliberately deferred"'s second `CostBasis` dimension + (`Charges []Charge`) decomposed by the same chain. +- **No `--policy` for `kilter domains`, no `kilter rds` verb.** RDS is reached + through `kilter domains --domain rds`, like every other domain. +- **`pkg/rds`'s `StorageParity` seam is still nil**, as U11 shipped it — + `--rds-fixture` cannot enable it and no flag pretends to. U13 owns it. + +--- + +## 7. Determinism + +- **Sort before summing money, proved by shuffle.** + `TestRDSOutputIsShuffleInvariantAndByteIdentical` permutes the recorded + account's instance pages and cluster list three ways and requires a + byte-identical report, then repeats the same run eight times *in one process* + (Go randomizes map iteration on every `range`, so in-process repetition is the + real test). `TestWhyCostIsOrderAndRepeatIndependent` does the same for + snapshot order on the command line. + `TestBacktestOutputIsByteIdenticalAcrossRuns` covers the scorecard, text and + JSON, on a noisy trace. +- **No map iteration reaches output.** `basisFrom` sorts its node groups and + namespaces before appending — they are summed downstream. `writeScorecard` + sorts the refusal codes. `loadLedgerActions` sorts by `(At, Fingerprint)`, + because a ledger read in file order would make the attribution depend on the + order entries happened to be appended. +- **No clock in any decision.** `--now` in `kilter domains`; + `backtestEpoch` is a package constant, because a replay window that drifts + with wall-clock time makes two runs over the same configuration disagree; + `--from`/`--to` are required for `why-cost` and resolved to concrete + timestamps before anything is computed for `explain`, then echoed in the + output. +- **`loadSnapshotSeries` rejects duplicate timestamps**, for the reason + `pkg/backtest` rejects them: a tie has no defined replay order. +- **No live AWS or cluster call anywhere**, including tests. `--rds-fixture` + replays a recorded account through the real seams; `--kube-snapshot` reads + recorded cluster snapshots; `--demo` generates a trace. No credential is read + and `~/.aws` is never opened. + +## 8. Fixtures + +`cmd/kilter/testdata/rds-account.json` (39 KB) is generated by +`cmd/kilter/rdsfixture_test.go` and committed, and `TestWriteRDSFixture` asserts +the committed bytes still match the generator — the same contract-test idiom +`TestWriteDomainFixtures` already uses, so a diff means `pkg/rds`'s collector +changed and a reviewer should look. + +``` +go test ./cmd/kilter -run TestWriteRDSFixture -update-fixtures +``` + +Each of the seven instances is the interesting case, not the easy one: + +| Instance | Why it is there | +|---|---| +| `db-pg-primary` | PostgreSQL, 500 GiB gp2, autoscaling on, 80 % never used — trap 9 (page cache) and trap 8 (a floor with a dollar and no API) | +| `db-mysql-multiaz` | the same-shaped `FreeableMemory` series on MySQL, where it IS readable and the downsize is still refused — the trap nothing else in the tree catches | +| `db-pg-replica` | zero connections across the whole window: the one replica finding safe to state | +| `db-mssql` | SQL Server EE license-included — refused **by name** rather than quoted an open-source rate | +| `db-aurora` | Aurora, refused by name (trap 16) | +| `db-mysql-cluster` | a Multi-AZ DB **cluster** member on MySQL, refused under its own name and not Aurora's (§5.3 — the reason `DescribeDBClusters` is read at all) | +| `db-legacy` | `kilter.dev/mode=off` — the guardrail, reachable only because `ListTagsForResource` is wired | + +The metric cadence is deliberately coarse (one point per 6 hours). Every +evidence gate in `pkg/rds` is about window **span** and delivery completeness, +not sample count, so a dense series would add a megabyte of JSON and prove +nothing extra. The `why-cost` and `explain` fixtures are generated into +`t.TempDir()` rather than committed, for the same reason in reverse: they are +small, and the domain fixtures already carry the contract-test role. diff --git a/cmd/kilter/backtest.go b/cmd/kilter/backtest.go new file mode 100644 index 0000000..d18d776 --- /dev/null +++ b/cmd/kilter/backtest.go @@ -0,0 +1,527 @@ +package main + +import ( + "encoding/json" + "flag" + "fmt" + "io" + "os" + "sort" + "strconv" + "strings" + "time" + + "github.com/agenticode/kilter/pkg/backtest" + "github.com/agenticode/kilter/pkg/decision" + "github.com/agenticode/kilter/pkg/plan" + "github.com/agenticode/kilter/pkg/recommend" +) + +// `kilter backtest` — the falsifiability harness, made runnable. +// +// pkg/backtest turns twelve domains' worth of self-assertion into a number, +// and on its first run it falsified the shipped engine. Until now nothing in +// the binary could call it. +// +// # What is wired, and the one thing that is not +// +// The TRACE path is wired: four synthetic archetypes whose oracles are known +// in closed form, replayed through the same observe-then-ask sequence +// pkg/api's Ingest/Plan runs. Policy files, the A/B comparison, Gate, the +// JSON scorecard and the CI exit code all work over it. +// +// The LIVE path is NOT wired, and refuses rather than pretending. §4.4's +// headline feature is "replay this cluster's own history", and it needs a seam +// that does not exist: pkg/store keeps only the LATEST snapshot per cluster +// (SaveSnapshot/LoadSnapshot are keyed by cluster, not by time), and +// recommend.ObserveSnapshot and plan.Build both take a *model.ClusterSnapshot, +// so a replay through the production code path needs topology over time. +// +// Running the harness against a single snapshot would produce a scorecard — +// one instant, no horizon, every number a rounding artefact — that looks +// exactly like a real one. That is the failure mode this command refuses to +// have: --cluster names the missing seam and exits non-zero. See +// backtestLiveRefusal. + +const backtestUsage = `kilter backtest — replay a policy against history and score it + +Usage: + kilter backtest --demo [flags] score the shipped policy over a synthetic trace + kilter backtest --cluster [flags] replay a cluster's own history (see below) + +Archetypes (--demo), each with a closed-form oracle: + steady every sample at the base level + diurnal half the window at peak + bursty 12 spikes/day — narrower than the 5%% CPU tail, wide enough for the memory peak + regime-change a level shift at the midpoint, on a decision instant + +Flags: + --demo KIND synthetic archetype to score + --cluster ID replay a live cluster's history (refused: see --help output) + --days N trace length in days (default 7) + --workloads N containers in the trace (default 2) + --noise PCT deterministic jitter, e.g. 0.05 for +/-5%% (default 0) + --horizon DUR how far ahead each decision is scored (default 24h) + --interval DUR spacing of decision instants (default 24h) + --starvation F CPU violation threshold: future p95 > request x F (default 1.0) + --incident-usd N price of one violated container-window (default 50) + --derive-costs derive CPU/memory rates from the catalog and the trace's nodes + --catalog PATH pricing catalog JSON (default: embedded) + --policy PATH policy triple JSON; omitted means the shipped default + --compare PATH score a second policy and print Gate's verdict + --enforce-refusals run pkg/decision's refusal predicates (models the pending wiring); + a policy file's "enforceDecisionRefusals" overrides it per policy + --json emit the scorecard verbatim (byte-stable, CI-diffable) + --fail-on-regression exit non-zero when Gate rejects the candidate +` + +func runBacktest(args []string) error { return runBacktestTo(os.Stdout, args) } + +// backtestFlags is the command's input. +type backtestFlags struct { + demo string + cluster string + days int + workloads int + noise float64 + + horizon time.Duration + interval time.Duration + starvation float64 + incidentUSD float64 + deriveCosts bool + catalog string + + policy string + compare string + enforceRefusals bool + jsonOut bool + failOnRegress bool +} + +// runBacktestTo is the testable entry point: everything printed goes to w. +func runBacktestTo(w io.Writer, args []string) error { + fs := flag.NewFlagSet("backtest", flag.ContinueOnError) + fs.SetOutput(w) + var bf backtestFlags + fs.StringVar(&bf.demo, "demo", "", "synthetic archetype (steady|diurnal|bursty|regime-change)") + fs.StringVar(&bf.cluster, "cluster", "", "replay a live cluster's own history") + fs.IntVar(&bf.days, "days", 7, "trace length in days") + fs.IntVar(&bf.workloads, "workloads", 2, "containers in the trace") + fs.Float64Var(&bf.noise, "noise", 0, "deterministic jitter fraction") + fs.DurationVar(&bf.horizon, "horizon", 24*time.Hour, "scoring horizon") + fs.DurationVar(&bf.interval, "interval", 24*time.Hour, "decision interval") + fs.Float64Var(&bf.starvation, "starvation", 0, "CPU starvation factor (0 = package default)") + fs.Float64Var(&bf.incidentUSD, "incident-usd", 0, "price of one violated container-window") + fs.BoolVar(&bf.deriveCosts, "derive-costs", false, "derive cost rates from the catalog") + fs.StringVar(&bf.catalog, "catalog", "", "pricing catalog JSON") + fs.StringVar(&bf.policy, "policy", "", "policy triple JSON") + fs.StringVar(&bf.compare, "compare", "", "second policy triple JSON to compare against") + fs.BoolVar(&bf.enforceRefusals, "enforce-refusals", false, "run pkg/decision's refusal predicates") + fs.BoolVar(&bf.jsonOut, "json", false, "emit the scorecard as JSON") + fs.BoolVar(&bf.failOnRegress, "fail-on-regression", false, "exit non-zero when Gate rejects") + if err := fs.Parse(args); err != nil { + return err + } + + switch { + case bf.cluster != "" && bf.demo != "": + return fmt.Errorf("backtest: --demo and --cluster are different sources of history; pass one") + case bf.cluster != "": + return backtestLiveRefusal(bf.cluster) + case bf.demo == "": + fmt.Fprint(w, backtestUsage) + return fmt.Errorf("backtest: --demo or --cluster is required") + } + return runBacktestDemo(w, &bf) +} + +// backtestLiveRefusal is the honest half of this command. +// +// It names the seam, says what it would take, and exits non-zero. It does NOT +// fall back to the one snapshot pkg/store holds: a scorecard computed over a +// single instant has the same shape, the same field names and the same +// confident tone as a real one, and an operator reading "regret $0.00" has no +// way to tell that it means "nothing was replayed". +func backtestLiveRefusal(cluster string) error { + return fmt.Errorf(`backtest --cluster %s: refused — snapshot history is not persisted. + +pkg/store keeps only the LATEST snapshot per cluster (SaveSnapshot/LoadSnapshot +are keyed by cluster, not by time), and pkg/evidence stores per-subject usage +series and events rather than pod/node topology. A replay that goes through the +production code path needs topology over time, because recommend.ObserveSnapshot +and plan.Build both take a *model.ClusterSnapshot. + +Scoring the one snapshot that does exist would produce a scorecard that looks +exactly like a real one, so this refuses instead. + +What it needs (pkg/backtest/FINDINGS.md, "Seams this unit needed and did not +find", item 1): a time-keyed snapshot bucket in pkg/store — SaveSnapshotAt(snap) +and Snapshots(cluster, from, to) — bounded the way the rest of the substrate is, +plus an adapter implementing backtest.SnapshotSource. At a 5-minute cadence a +30-day window is 8,640 snapshots per cluster, so the realistic shape is a +keyframe-plus-delta encoding or a reduced replay snapshot carrying only what +recommend and plan read. + +Meanwhile: kilter backtest --demo regime-change`, cluster) +} + +// runBacktestDemo scores one or two policies over a synthetic trace. +func runBacktestDemo(w io.Writer, bf *backtestFlags) error { + kind, err := parseArchetype(bf.demo) + if err != nil { + return err + } + // The trace's start is a FIXED instant, never time.Now(): the replay + // window must come from the history, or two runs over the same + // configuration disagree. + spec := backtest.TraceSpec{ + Cluster: "demo-" + string(kind), + Kind: kind, + Start: backtestEpoch, + Days: bf.days, + Workloads: bf.workloads, + NoisePct: bf.noise, + } + trace, err := spec.Build() + if err != nil { + return fmt.Errorf("backtest --demo %s: %w", bf.demo, err) + } + store, err := trace.Store() + if err != nil { + return fmt.Errorf("backtest: build evidence store: %w", err) + } + + scoring := backtest.DefaultConfig() + scoring.DecisionInterval = bf.interval + if bf.starvation > 0 { + scoring.StarvationFactor = bf.starvation + } + if bf.incidentUSD > 0 { + scoring.Cost.IncidentUSD = bf.incidentUSD + } + catalog, err := loadCatalog(bf.catalog) + if err != nil { + return err + } + if bf.deriveCosts { + if len(trace.Snapshots) == 0 { + return fmt.Errorf("backtest: --derive-costs needs at least one snapshot") + } + cm, err := backtest.CostModelFromCatalog(catalog, trace.Snapshots[len(trace.Snapshots)-1], scoring.Cost) + if err != nil { + return fmt.Errorf("backtest: --derive-costs: %w", err) + } + scoring.Cost = cm + } + + base, err := loadPolicy(bf.policy, bf.enforceRefusals) + if err != nil { + return err + } + run := func(p policy) (*backtest.Scorecard, error) { + h := &backtest.Harness{ + Evidence: store, + History: trace.Source(), + Rec: p.Rec, + Plan: p.Plan, + Decision: p.Decision, + EnforceDecisionRefusals: p.EnforceRefusals, + Catalog: catalog, + Scoring: scoring, + } + return h.Run(trace.Cluster, trace.Start, trace.End, bf.horizon) + } + + current, err := run(base) + if err != nil { + return fmt.Errorf("backtest: %w", err) + } + + var candidate *backtest.Scorecard + var gateOK bool + var gateReasons []string + if bf.compare != "" { + alt, err := loadPolicy(bf.compare, bf.enforceRefusals) + if err != nil { + return err + } + candidate, err = run(alt) + if err != nil { + return fmt.Errorf("backtest --compare: %w", err) + } + gateOK, gateReasons = backtest.Gate(current, candidate, backtest.DefaultTolerance()) + } + + if bf.jsonOut { + if candidate == nil { + // Verbatim: Scorecard.Encode is the byte-stable, CI-diffable form. + raw, err := current.Encode() + if err != nil { + return err + } + _, err = w.Write(append(raw, '\n')) + return failOnGate(err, bf, candidate, gateOK) + } + out := map[string]any{ + "current": current, "candidate": candidate, + "gate": map[string]any{"accepted": gateOK, "reasons": gateReasons}, + } + if err := writeJSON(w, out); err != nil { + return err + } + return failOnGate(nil, bf, candidate, gateOK) + } + + var b strings.Builder + fmt.Fprintf(&b, "kilter backtest — %s trace, %d days, %d workloads\n", + kind, bf.days, bf.workloads) + fmt.Fprintf(&b, "window %s .. %s horizon %s interval %s\n\n", + trace.Start.UTC().Format(time.RFC3339), trace.End.UTC().Format(time.RFC3339), + bf.horizon, scoring.DecisionInterval) + writeScorecard(&b, "policy", current) + if candidate != nil { + b.WriteString("\n") + writeScorecard(&b, "candidate", candidate) + b.WriteString("\n Gate\n") + if gateOK { + b.WriteString(" ACCEPTED: the candidate dominates on the terms §4.6 defines\n") + } else { + b.WriteString(" REJECTED\n") + } + for _, r := range gateReasons { + fmt.Fprintf(&b, " %s\n", r) + } + } + if _, err := io.WriteString(w, b.String()); err != nil { + return err + } + return failOnGate(nil, bf, candidate, gateOK) +} + +// failOnGate turns a rejected comparison into the CI exit code §4.4 asks for. +func failOnGate(prior error, bf *backtestFlags, candidate *backtest.Scorecard, ok bool) error { + if prior != nil { + return prior + } + if bf.failOnRegress && candidate != nil && !ok { + return fmt.Errorf("backtest: the candidate policy did not pass the gate") + } + return nil +} + +// writeScorecard renders one scorecard. Every enumeration is sorted, so two +// runs over the same trace print the same bytes. +func writeScorecard(b *strings.Builder, label string, s *backtest.Scorecard) { + fmt.Fprintf(b, " %s %s\n", label, s.Policy) + fmt.Fprintf(b, " scored %d decisions %d refusals %d (good %d, idle %d)\n", + s.Scored, s.Decisions, s.Scored-s.Decisions, s.RefusalsGood, s.RefusalsIdle) + fmt.Fprintf(b, " safety memViolations %d cpuStarvation %d oomKills %d\n", + s.MemViolations, s.CPUStarvation, s.MemOOMKills) + // OracleGapPct is ALREADY scaled by 100 inside pkg/backtest (scorecard.go: + // `meanSorted(gaps) * 100`), even though the doc comment states the + // unscaled ratio. Multiplying again here printed a 9,445 % oracle gap for + // the regime-change golden whose real value is 94.5. + fmt.Fprintf(b, " efficiency oracleGap %.1f%% (applied %.1f%%) claimed/realized %.2f\n", + s.OracleGapPct, s.OracleGapPctApplied, s.ClaimedVsRealized) + fmt.Fprintf(b, " stability flipRate %.3f flips %d\n", s.FlipRate, s.Flips) + fmt.Fprintf(b, " regret $%.2f (resource $%.2f + risk $%.2f)\n", + s.RegretUSD, s.ResourceRegretUSD, s.RiskRegretUSD) + fmt.Fprintf(b, " forgone $%.2f left on the table by idle refusals\n", s.ForgoneSavingsUSD) + if len(s.Refusals) > 0 { + codes := make([]string, 0, len(s.Refusals)) + for c := range s.Refusals { + codes = append(codes, c) + } + sort.Strings(codes) + b.WriteString(" why it refused\n") + for _, c := range codes { + fmt.Fprintf(b, " %-28s %d\n", c, s.Refusals[c]) + } + } +} + +// backtestEpoch is the trace's fixed start. It is a constant rather than a +// clock read because a replay window that drifts with wall-clock time makes +// two runs over the same configuration disagree — and the whole value of a +// scorecard is that it is comparable. +var backtestEpoch = time.Date(2026, 1, 5, 0, 0, 0, 0, time.UTC) + +func parseArchetype(s string) (backtest.TraceKind, error) { + k := backtest.TraceKind(strings.TrimSpace(s)) + for _, known := range []backtest.TraceKind{ + backtest.TraceSteady, backtest.TraceDiurnal, + backtest.TraceBursty, backtest.TraceRegimeChange, + } { + if k == known { + return k, nil + } + } + return "", fmt.Errorf("backtest --demo %q: unknown archetype (known: steady, diurnal, bursty, regime-change)", s) +} + +// ---------------------------------------------------------------- policy file + +// policy is the (recommend, plan, decision) triple under test. +type policy struct { + Rec recommend.Config + Plan plan.Config + Decision decision.Config + // EnforceRefusals runs pkg/decision's refusal predicates before a + // recommendation is accepted. It is part of the POLICY rather than of the + // scoring knobs because it is pending production wiring, not a yardstick: + // pkg/decision shipped as unit 3 and pkg/recommend does not import it, so + // today a refusal predicate cannot stop a recommendation from being + // planned. Making it a policy field is what turns "should we wire the + // decision layer in?" into an A/B through Gate instead of an opinion. + EnforceRefusals bool +} + +// policyFile is the on-disk shape of a policy triple. +// +// It is a cmd/-side projection rather than the three Config structs directly, +// and the mismatch that forced it is worth naming: none of recommend.Config, +// plan.Config or decision.Config carries json tags, so decoding into them +// gives a file whose keys are Go field names and whose durations are integer +// nanoseconds — and, worse, whose OMITTED fields decode as zero. A policy file +// where leaving out `cpuHeadroom` silently means "headroom 1.0" is a footgun +// with a business consequence. Every field here is therefore a POINTER, +// overlaid onto the package defaults, so absent means default and present +// means present. +type policyFile struct { + Recommend *struct { + CPUPercentile *float64 `json:"cpuPercentile,omitempty"` + CPUHeadroom *float64 `json:"cpuHeadroom,omitempty"` + MemoryPercentile *float64 `json:"memoryPercentile,omitempty"` + MemoryHeadroom *float64 `json:"memoryHeadroom,omitempty"` + MinSamples *int `json:"minSamples,omitempty"` + MinWindow *string `json:"minWindow,omitempty"` + MinChangeRatio *float64 `json:"minChangeRatio,omitempty"` + SkipCPUForHPA *bool `json:"skipCPUForHPA,omitempty"` + } `json:"recommend,omitempty"` + Plan *struct { + MinNodeUtilization *float64 `json:"minNodeUtilization,omitempty"` + MinConfidence *float64 `json:"minConfidence,omitempty"` + MaxNodeRemovals *int `json:"maxNodeRemovals,omitempty"` + ApplyRecommendations *bool `json:"applyRecommendations,omitempty"` + MinClusterHeadroom *float64 `json:"minClusterHeadroom,omitempty"` + RespectManagedNodes *bool `json:"respectManagedNodes,omitempty"` + DefaultMode *string `json:"defaultMode,omitempty"` + } `json:"plan,omitempty"` + Decision *struct { + MinSamples *int `json:"minSamples,omitempty"` + MinWindow *string `json:"minWindow,omitempty"` + BaseSoak *string `json:"baseSoak,omitempty"` + ClassFlipWindow *string `json:"classFlipWindow,omitempty"` + MinClassStability *float64 `json:"minClassStability,omitempty"` + MaxHPAThrashPerHour *float64 `json:"maxHPAThrashPerHour,omitempty"` + MaxForecastDivergence *float64 `json:"maxForecastDivergence,omitempty"` + ActConfidence *float64 `json:"actConfidence,omitempty"` + } `json:"decision,omitempty"` + // EnforceDecisionRefusals models the pending pkg/decision wiring. See + // policy.EnforceRefusals. + EnforceDecisionRefusals *bool `json:"enforceDecisionRefusals,omitempty"` +} + +// loadPolicy reads a policy triple, defaulting to the shipped policy. +func loadPolicy(path string, enforceDefault bool) (policy, error) { + p := policy{ + Rec: recommend.DefaultConfig(), + Plan: plan.DefaultConfig(), + Decision: decision.DefaultConfig(), + } + p.EnforceRefusals = enforceDefault + if path == "" || path == "default" { + return p, nil + } + raw, err := os.ReadFile(path) + if err != nil { + return policy{}, fmt.Errorf("--policy: %w", err) + } + var pf policyFile + dec := json.NewDecoder(strings.NewReader(string(raw))) + // Unknown fields are rejected: a knob misspelled in a policy file that is + // silently ignored produces a scorecard for a policy nobody ran. + dec.DisallowUnknownFields() + if err := dec.Decode(&pf); err != nil { + return policy{}, fmt.Errorf("%s: %w", path, err) + } + if r := pf.Recommend; r != nil { + setF(&p.Rec.CPUPercentile, r.CPUPercentile) + setF(&p.Rec.CPUHeadroom, r.CPUHeadroom) + setF(&p.Rec.MemoryPercentile, r.MemoryPercentile) + setF(&p.Rec.MemoryHeadroom, r.MemoryHeadroom) + setI(&p.Rec.MinSamples, r.MinSamples) + if err := setD(&p.Rec.MinWindow, r.MinWindow, path, "recommend.minWindow"); err != nil { + return policy{}, err + } + setF(&p.Rec.MinChangeRatio, r.MinChangeRatio) + setB(&p.Rec.SkipCPUForHPA, r.SkipCPUForHPA) + } + if pl := pf.Plan; pl != nil { + setF(&p.Plan.MinNodeUtilization, pl.MinNodeUtilization) + setF(&p.Plan.MinConfidence, pl.MinConfidence) + setI(&p.Plan.MaxNodeRemovals, pl.MaxNodeRemovals) + setB(&p.Plan.ApplyRecommendations, pl.ApplyRecommendations) + setF(&p.Plan.MinClusterHeadroom, pl.MinClusterHeadroom) + setB(&p.Plan.RespectManagedNodes, pl.RespectManagedNodes) + setS(&p.Plan.DefaultMode, pl.DefaultMode) + } + if d := pf.Decision; d != nil { + setI(&p.Decision.MinSamples, d.MinSamples) + if err := setD(&p.Decision.MinWindow, d.MinWindow, path, "decision.minWindow"); err != nil { + return policy{}, err + } + if err := setD(&p.Decision.BaseSoak, d.BaseSoak, path, "decision.baseSoak"); err != nil { + return policy{}, err + } + if err := setD(&p.Decision.ClassFlipWindow, d.ClassFlipWindow, path, "decision.classFlipWindow"); err != nil { + return policy{}, err + } + setF(&p.Decision.MinClassStability, d.MinClassStability) + setF(&p.Decision.MaxHPAThrashPerHour, d.MaxHPAThrashPerHour) + setF(&p.Decision.MaxForecastDivergence, d.MaxForecastDivergence) + setF(&p.Decision.ActConfidence, d.ActConfidence) + } + setB(&p.EnforceRefusals, pf.EnforceDecisionRefusals) + return p, nil +} + +func setF(dst *float64, src *float64) { + if src != nil { + *dst = *src + } +} +func setI(dst *int, src *int) { + if src != nil { + *dst = *src + } +} +func setB(dst *bool, src *bool) { + if src != nil { + *dst = *src + } +} +func setS(dst *string, src *string) { + if src != nil { + *dst = *src + } +} + +// setD parses a duration written the way a human writes one ("6h", "45m"). +func setD(dst *time.Duration, src *string, path, field string) error { + if src == nil { + return nil + } + d, err := time.ParseDuration(*src) + if err != nil { + // A bare number is a common mistake and means nanoseconds in Go's + // encoding, which is never what anybody meant. + if _, numErr := strconv.Atoi(*src); numErr == nil { + return fmt.Errorf("%s: %s = %q: durations need a unit (\"6h\", \"45m\")", path, field, *src) + } + return fmt.Errorf("%s: %s: %w", path, field, err) + } + *dst = d + return nil +} diff --git a/cmd/kilter/backtest_test.go b/cmd/kilter/backtest_test.go new file mode 100644 index 0000000..374866b --- /dev/null +++ b/cmd/kilter/backtest_test.go @@ -0,0 +1,291 @@ +package main + +import ( + "encoding/json" + "os" + "path/filepath" + "strings" + "testing" + + "github.com/agenticode/kilter/pkg/backtest" +) + +// These tests drive pkg/backtest's REAL harness through the REAL CLI entry +// point. The traces are synthetic and the oracles are known in closed form, +// so every number below is reproducible: there is no clock in the command +// (backtestEpoch is a constant) and no network anywhere. + +func runBacktestOK(t *testing.T, args ...string) string { + t.Helper() + var b strings.Builder + if err := runBacktestTo(&b, args); err != nil { + t.Fatalf("kilter backtest %s: %v\n%s", strings.Join(args, " "), err, b.String()) + } + return b.String() +} + +func backtestScorecard(t *testing.T, args ...string) *backtest.Scorecard { + t.Helper() + raw := runBacktestOK(t, append(args, "--json")...) + var sc backtest.Scorecard + if err := json.Unmarshal([]byte(raw), &sc); err != nil { + t.Fatalf("decode scorecard: %v\n%s", err, raw) + } + return &sc +} + +func writePolicy(t *testing.T, body string) string { + t.Helper() + path := filepath.Join(t.TempDir(), "policy.json") + if err := os.WriteFile(path, []byte(body), 0o644); err != nil { + t.Fatal(err) + } + return path +} + +// TestBacktestReproducesTheShippedGoldens. +// +// pkg/backtest/FINDINGS.md publishes what its goldens say about the shipped +// engine over 7 days and 2 workloads. Those numbers were produced inside the +// package; this asserts the CLI reaches the same ones, which is the difference +// between a harness that exists and a harness a user can run. +func TestBacktestReproducesTheShippedGoldens(t *testing.T) { + for _, tc := range []struct { + archetype string + decisions int + mem, cpu int + gapApplied float64 + flipRate float64 + regret float64 + }{ + {"steady", 12, 0, 0, 17.3, 0, 2.74}, + {"diurnal", 12, 0, 0, 40.5, 0, 4.02}, + {"bursty", 12, 0, 0, 23.4, 0, 2.90}, + {"regime-change", 12, 2, 2, 5.7, 0.167, 202.57}, + } { + t.Run(tc.archetype, func(t *testing.T) { + sc := backtestScorecard(t, "--demo", tc.archetype) + if sc.Decisions != tc.decisions { + t.Errorf("decisions = %d, want %d", sc.Decisions, tc.decisions) + } + if sc.MemViolations != tc.mem || sc.CPUStarvation != tc.cpu { + t.Errorf("safety = (mem %d, cpu %d), want (%d, %d)", + sc.MemViolations, sc.CPUStarvation, tc.mem, tc.cpu) + } + if !near(sc.OracleGapPctApplied, tc.gapApplied, 0.05) { + t.Errorf("applied oracle gap = %.2f, want %.1f", sc.OracleGapPctApplied, tc.gapApplied) + } + if !near(sc.FlipRate, tc.flipRate, 0.001) { + t.Errorf("flip rate = %.3f, want %.3f", sc.FlipRate, tc.flipRate) + } + if !near(sc.RegretUSD, tc.regret, 0.005) { + t.Errorf("regret = $%.2f, want $%.2f", sc.RegretUSD, tc.regret) + } + }) + } +} + +// TestOracleGapIsRenderedInTheUnitsTheScorecardUses. +// +// Regression for a real bug in this wiring: Scorecard.OracleGapPct is ALREADY +// multiplied by 100 inside pkg/backtest, while its doc comment states the +// unscaled ratio. The first draft of the renderer multiplied again and printed +// a 9,445 % oracle gap for a trace whose real gap is 94.5 %. +func TestOracleGapIsRenderedInTheUnitsTheScorecardUses(t *testing.T) { + sc := backtestScorecard(t, "--demo", "regime-change") + out := runBacktestOK(t, "--demo", "regime-change") + want := strings.TrimSpace(strings.Split(strings.TrimSpace( + strings.Replace(strings.Split(out, "oracleGap ")[1], "%", " ", 1)), " ")[0]) + if want != "94.5" { + t.Fatalf("rendered oracle gap %q, want 94.5", want) + } + if !near(sc.OracleGapPct, 94.5, 0.05) { + t.Fatalf("scorecard oracle gap = %v, want ~94.5", sc.OracleGapPct) + } +} + +// TestBacktestLiveHistoryRefusesRatherThanScoringOneSnapshot. +// +// This is the honest half of the command. pkg/store keeps only the LATEST +// snapshot per cluster, so a live replay has no history to replay. Running the +// harness against that one snapshot would yield a scorecard with the same +// shape, the same field names and the same confident tone as a real one — the +// worst possible failure, because the number looks fine. +func TestBacktestLiveHistoryRefusesRatherThanScoringOneSnapshot(t *testing.T) { + var b strings.Builder + err := runBacktestTo(&b, []string{"--cluster", "prod"}) + if err == nil { + t.Fatalf("a live backtest was accepted:\n%s", b.String()) + } + msg := err.Error() + for _, want := range []string{ + "snapshot history is not persisted", + "pkg/store", + "SaveSnapshotAt", + "Snapshots(cluster, from, to)", + "backtest.SnapshotSource", + "--demo", + } { + if !strings.Contains(msg, want) { + t.Errorf("the refusal does not mention %q:\n%s", want, msg) + } + } + if strings.Contains(b.String(), "regret") { + t.Errorf("a scorecard was printed for a cluster with no history:\n%s", b.String()) + } + // And it refuses rather than quietly preferring one source over the other. + if err := runBacktestTo(&b, []string{"--cluster", "prod", "--demo", "steady"}); err == nil { + t.Error("--cluster and --demo together were accepted") + } +} + +// TestWiringTheDecisionLayerIsAnImprovementThroughTheCLI. +// +// pkg/backtest's headline result, reproduced end to end: on the regime-change +// trace, enforcing pkg/decision's refusal predicates removes ALL CPU +// starvation and halves total regret, bought with a few dollars of extra idle +// headroom. The test asserts the win AND that it was paid for — a scorecard +// that reported only the win would be an advertisement. +func TestWiringTheDecisionLayerIsAnImprovementThroughTheCLI(t *testing.T) { + enforced := writePolicy(t, `{"enforceDecisionRefusals": true}`) + raw := runBacktestOK(t, "--demo", "regime-change", "--workloads", "3", + "--compare", enforced, "--json") + var env struct { + Current *backtest.Scorecard `json:"current"` + Candidate *backtest.Scorecard `json:"candidate"` + Gate struct { + Accepted bool `json:"accepted"` + Reasons []string `json:"reasons"` + } `json:"gate"` + } + if err := json.Unmarshal([]byte(raw), &env); err != nil { + t.Fatalf("decode: %v\n%s", err, raw) + } + cur, cand := env.Current, env.Candidate + + if cur.CPUStarvation == 0 { + t.Fatal("the shipped policy starved nothing; the A/B proves nothing") + } + if cand.CPUStarvation != 0 { + t.Errorf("enforced refusals left %d CPU starvations, want 0", cand.CPUStarvation) + } + if cand.RegretUSD >= cur.RegretUSD/1.5 { + t.Errorf("regret $%.2f -> $%.2f: expected roughly a halving", cur.RegretUSD, cand.RegretUSD) + } + if cand.ResourceRegretUSD <= cur.ResourceRegretUSD { + t.Errorf("resource regret $%.2f -> $%.2f: the safety win must be PAID for, "+ + "and the scorecard must report both halves of the trade", + cur.ResourceRegretUSD, cand.ResourceRegretUSD) + } + if cur.MemViolations != cand.MemViolations { + t.Errorf("memory violations moved (%d -> %d); the level-shift window is "+ + "unavoidable and no policy should change it", cur.MemViolations, cand.MemViolations) + } + if !env.Gate.Accepted { + t.Errorf("Gate rejected a strict improvement: %v", env.Gate.Reasons) + } + if _, ok := cand.Refusals["post-change-soak"]; !ok { + t.Errorf("no post-change-soak refusal fired: %v", cand.Refusals) + } +} + +// TestFailOnRegressionIsTheCIGate: a policy that refuses everything banks no +// savings and still pays for whatever the unchanged sizing did, so it cannot +// dominate — and --fail-on-regression turns that into an exit code. +func TestFailOnRegressionIsTheCIGate(t *testing.T) { + refuser := writePolicy(t, `{"recommend": {"minSamples": 1000000}}`) + + // Without the flag the comparison is reported and the command succeeds. + out := runBacktestOK(t, "--demo", "steady", "--compare", refuser) + if !strings.Contains(out, "REJECTED") { + t.Fatalf("Gate accepted a refuse-everything policy:\n%s", out) + } + + var b strings.Builder + err := runBacktestTo(&b, []string{"--demo", "steady", "--compare", refuser, "--fail-on-regression"}) + if err == nil { + t.Fatal("--fail-on-regression did not fail on a rejected candidate") + } + if !strings.Contains(b.String(), "REJECTED") { + t.Errorf("the reasons were not printed alongside the failure:\n%s", b.String()) + } +} + +// TestBacktestOutputIsByteIdenticalAcrossRuns. Go randomizes map iteration on +// every range, so repeating in ONE process is the real determinism test. +func TestBacktestOutputIsByteIdenticalAcrossRuns(t *testing.T) { + base := runBacktestOK(t, "--demo", "bursty", "--noise", "0.05") + baseJSON := runBacktestOK(t, "--demo", "bursty", "--noise", "0.05", "--json") + for i := 0; i < 6; i++ { + if got := runBacktestOK(t, "--demo", "bursty", "--noise", "0.05"); got != base { + t.Fatalf("text run %d differs", i) + } + if got := runBacktestOK(t, "--demo", "bursty", "--noise", "0.05", "--json"); got != baseJSON { + t.Fatalf("json run %d differs", i) + } + } +} + +// TestPolicyFileFailsLoudly. +// +// A knob misspelled in a policy file that is silently ignored produces a +// scorecard for a policy nobody ran — and the scorecard looks fine. Unknown +// fields are rejected; so is a bare number where a duration belongs, because +// Go would read it as nanoseconds and nobody ever meant that. +func TestPolicyFileFailsLoudly(t *testing.T) { + for _, tc := range []struct{ name, body, want string }{ + {"unknown field", `{"recommend": {"cpuHeadrooom": 1.5}}`, "unknown field"}, + {"bare duration", `{"decision": {"baseSoak": "3600"}}`, "durations need a unit"}, + {"bad duration", `{"recommend": {"minWindow": "six hours"}}`, "minWindow"}, + {"not json", `{`, "unexpected EOF"}, + } { + t.Run(tc.name, func(t *testing.T) { + path := writePolicy(t, tc.body) + var b strings.Builder + err := runBacktestTo(&b, []string{"--demo", "steady", "--policy", path}) + if err == nil { + t.Fatalf("a broken policy file was accepted:\n%s", b.String()) + } + if !strings.Contains(err.Error(), tc.want) { + t.Errorf("error = %q, want it to mention %q", err, tc.want) + } + }) + } + // A missing knob means DEFAULT, not zero: an empty policy file must score + // exactly like no policy file at all. + empty := writePolicy(t, `{}`) + if runBacktestOK(t, "--demo", "steady", "--policy", empty) != runBacktestOK(t, "--demo", "steady") { + t.Error("an empty policy file scored differently from the shipped default") + } +} + +// TestBacktestRejectsUnknownArchetypes. +func TestBacktestRejectsUnknownArchetypes(t *testing.T) { + var b strings.Builder + if err := runBacktestTo(&b, []string{"--demo", "chaotic"}); err == nil || + !strings.Contains(err.Error(), "unknown archetype") { + t.Fatalf("err = %v, want an unknown-archetype error", err) + } + if err := runBacktestTo(&b, nil); err == nil { + t.Fatal("backtest with no source of history was accepted") + } +} + +// TestDerivedCostsComeFromTheCatalogNotTheDefaults. +func TestDerivedCostsComeFromTheCatalogNotTheDefaults(t *testing.T) { + def := backtestScorecard(t, "--demo", "steady") + derived := backtestScorecard(t, "--demo", "steady", "--derive-costs") + if derived.Cost == def.Cost { + t.Skip("the embedded catalog happens to agree with the default cost model") + } + if derived.Cost.IncidentUSD != def.Cost.IncidentUSD { + t.Errorf("--derive-costs moved IncidentUSD (%v -> %v); it prices RISK, "+ + "which no node catalog knows about", + def.Cost.IncidentUSD, derived.Cost.IncidentUSD) + } +} + +func near(got, want, tol float64) bool { + d := got - want + return d <= tol && -d <= tol +} diff --git a/cmd/kilter/domains.go b/cmd/kilter/domains.go index 4436cf7..7de59bf 100644 --- a/cmd/kilter/domains.go +++ b/cmd/kilter/domains.go @@ -1,11 +1,13 @@ package main import ( + "context" "encoding/json" "errors" "flag" "fmt" "io" + "math" "os" "sort" "strings" @@ -16,9 +18,11 @@ import ( domecs "github.com/agenticode/kilter/pkg/domain/ecs" "github.com/agenticode/kilter/pkg/domain/fargate" domlambda "github.com/agenticode/kilter/pkg/domain/lambda" + domrds "github.com/agenticode/kilter/pkg/domain/rds" "github.com/agenticode/kilter/pkg/guard" "github.com/agenticode/kilter/pkg/model" kcommit "github.com/agenticode/kilter/pkg/pricing/commit" + krds "github.com/agenticode/kilter/pkg/rds" ) // `kilter domains` is where eight packages of decision logic become reachable @@ -52,6 +56,9 @@ Input is recorded snapshots; this command makes no cloud call. Flags: --snapshot PATH domain snapshot JSON; repeatable, routed by its "domain" field --kube-snapshot PATH cluster snapshot JSON (kilter analyze --dump-snapshot) for k8s-fargate + --rds-fixture PATH recorded RDS account; runs the real rds collector (repeatable) + --rds-rates PATH RDS rate override JSON; layered over the shipped unverified table + --rds-window DUR RDS observation window (default 336h); clamped to CloudWatch retention --commitments PATH RI/Savings-Plan inventory JSON (kilter pricing sync-commitments) --catalog PATH pricing catalog JSON (default: embedded) --domain KIND restrict to one domain; repeatable (%s) @@ -59,6 +66,7 @@ Flags: --region REGION region label for commitment usage lines --now RFC3339 decision time (default: now) --json machine-readable output + --rds-detail also print pkg/rds's own refusals-first report plan-only flags: --max-steps N cap the plan @@ -107,7 +115,11 @@ func kindNames() []string { type domainFlags struct { snapshots repeatedFlag kubeSnaps repeatedFlag + rdsFixtures repeatedFlag kinds repeatedFlag + rdsRates string + rdsWindow time.Duration + rdsDetail bool commitments string catalog string scope string @@ -133,6 +145,10 @@ func (r *repeatedFlag) Set(v string) error { func (df *domainFlags) bind(fs *flag.FlagSet, withPlan bool) { fs.Var(&df.snapshots, "snapshot", "domain snapshot JSON (repeatable)") fs.Var(&df.kubeSnaps, "kube-snapshot", "cluster snapshot JSON for k8s-fargate (repeatable)") + fs.Var(&df.rdsFixtures, "rds-fixture", "recorded RDS account JSON, run through the real collector (repeatable)") + fs.StringVar(&df.rdsRates, "rds-rates", "", "RDS rate override JSON (pkg/rds LoadRates format)") + fs.DurationVar(&df.rdsWindow, "rds-window", 14*24*time.Hour, "RDS observation window") + fs.BoolVar(&df.rdsDetail, "rds-detail", false, "also print pkg/rds's own refusals-first report") fs.Var(&df.kinds, "domain", "restrict to one domain kind (repeatable)") fs.StringVar(&df.commitments, "commitments", "", "RI/Savings-Plan inventory JSON") fs.StringVar(&df.catalog, "catalog", "", "pricing catalog JSON (default: embedded)") @@ -159,6 +175,10 @@ type runtime struct { // registered lands here rather than being dropped: silently discarding // collected data is indistinguishable from a broken collector. Warnings []string + // rds is kept only so --rds-detail can render pkg/rds's own report, whose + // layout puts refusals first. Every other consumer goes through the + // registry. + rds *domrds.Domain } // buildRuntime wires the registry, feeds it every recorded snapshot, and @@ -228,6 +248,26 @@ func buildRuntime(df *domainFlags) (*runtime, error) { return nil, err } } + // RDS. It registers like any other domain and then refuses everything, + // which is the deliverable rather than a gap: the class is where the money + // is and changing it is a failover, allocated storage cannot shrink, and + // FreeableMemory is MemAvailable. Its Recommend() is empty by construction, + // so the whole output arrives through the Refuser seam. + var rdsDomain *domrds.Domain + if wanted[domain.RDS] { + card, err := loadRDSRates(df.rdsRates) + if err != nil { + return nil, err + } + d, err := domrds.New(domrds.Config{Scope: df.scope, Region: df.region, Rates: card}) + if err != nil { + return nil, err + } + if err := rt.Registry.Register(d); err != nil { + return nil, err + } + rdsDomain, rt.rds = d, d + } // Feed it. A snapshot that cannot be read is fatal (a path the operator // typed is wrong); a snapshot for a domain nobody registered is a warning @@ -268,7 +308,42 @@ func buildRuntime(df *domainFlags) (*runtime, error) { } } - rt.Ledger = buildLedger(rt.Registry, inv, now) + // The RDS collection loop (pkg/rds/FINDINGS.md §6.3), over a recorded + // account. Observe() takes the native snapshot rather than the generic + // projection because domain.Sample has no truncation flag: a series + // flattened into samples arrives looking complete, and a truncated + // DatabaseConnections series that looks complete is an idle verdict + // manufactured out of silence. + for _, path := range df.rdsFixtures { + if rdsDomain == nil { + rt.Warnings = append(rt.Warnings, + fmt.Sprintf("%s: RDS fixture supplied, but the rds domain is not registered here", path)) + continue + } + snap, warns, err := collectRDS(context.Background(), path, df.scope, df.region, now, df.rdsWindow) + if err != nil { + return nil, err + } + rt.Warnings = append(rt.Warnings, warns...) + if err := rdsDomain.Observe(snap); err != nil { + return nil, fmt.Errorf("%s: %w", path, err) + } + // §6.5: snap.Reservations is already []commit.ReservedDBInstance and + // goes straight into the account-wide inventory. An RDS line is + // absorbed by a Reserved DB Instance and by nothing else — no Savings + // Plan of any type covers RDS — so appending cannot disturb what + // --commitments contributed for the other domains. + if len(snap.Reservations) > 0 { + if inv == nil { + inv = &kcommit.Inventory{} + } + inv.ReservedDBs = append(inv.ReservedDBs, snap.Reservations...) + } + } + + ledger, ledgerWarnings := buildLedger(rt.Registry, inv, now) + rt.Ledger = ledger + rt.Warnings = append(rt.Warnings, ledgerWarnings...) sort.Strings(rt.Warnings) return rt, nil } @@ -287,17 +362,65 @@ func buildRuntime(df *domainFlags) (*runtime, error) { // what it currently costs — a figure that does not depend on any commitment, // because it is the on-demand rate of what is running today. The second nets // every domain's proposed change against that whole picture. -func buildLedger(reg *domain.Registry, inv *kcommit.Inventory, now time.Time) *domain.Ledger { +func buildLedger(reg *domain.Registry, inv *kcommit.Inventory, now time.Time) (*domain.Ledger, []string) { var lines []kcommit.UsageLine + var warnings []string + seen := map[string]domain.Kind{} for _, k := range reg.Kinds() { d, ok := reg.Get(k) if !ok { continue } - lines = append(lines, usageLinesOf(d, now)...) + for _, l := range usageLinesOf(d, now) { + if why, bad := badBaselineLine(l, seen); bad { + warnings = append(warnings, fmt.Sprintf( + "%s: dropped a usage line from the account-wide baseline (%s)", k, why)) + continue + } + seen[l.ID] = k + lines = append(lines, l) + } } sort.Slice(lines, func(i, j int) bool { return lines[i].ID < lines[j].ID }) - return domain.NewLedger(inv, kcommit.Usage{Lines: lines}) + return domain.NewLedger(inv, kcommit.Usage{Lines: lines}), warnings +} + +// badBaselineLine is the output check on the usage-line seam. +// +// It is the same hole j1-wire closed one level down. Registry.PlanSteps +// filtered its INPUT and not its output, so a domain could return a step +// labelled with another domain's name and borrow that domain's actuator; here +// a domain returns lines that go straight into the ACCOUNT-WIDE commitment +// baseline, which is what every OTHER domain's net savings are computed +// against. Three ways that goes wrong, all silent: +// +// - An EMPTY ID never matches in domain.Ledger's splice and is always +// appended, so two anonymous lines for one resource double-count the usage +// available to absorb a commitment — which OVERSTATES absorption and +// therefore overstates savings. +// - A DUPLICATE ID from a second domain replaces the first domain's line +// rather than adding to it, so one domain silently rewrites another's +// contribution. +// - A non-positive or non-finite rate or quantity prices real usage at +// nothing, which again makes a commitment look more absorbed than it is. +// +// Dropping is the conservative direction: fewer baseline lines means less +// usage to absorb a commitment, means more apparent stranding, means a +// smaller claimed saving. A drop is never silent — it lands in the collection +// warnings the CLI prints. +func badBaselineLine(l kcommit.UsageLine, seen map[string]domain.Kind) (string, bool) { + switch { + case l.ID == "": + return fmt.Sprintf("no ID (%s %s); an anonymous line cannot be spliced and would double-count", + l.Kind, l.InstanceType), true + case seen[l.ID] != "": + return fmt.Sprintf("ID %q already contributed by domain %q", l.ID, seen[l.ID]), true + case math.IsNaN(l.ODRate) || math.IsInf(l.ODRate, 0) || l.ODRate <= 0: + return fmt.Sprintf("ID %q has rate %v", l.ID, l.ODRate), true + case math.IsNaN(l.Quantity) || math.IsInf(l.Quantity, 0) || l.Quantity <= 0: + return fmt.Sprintf("ID %q has quantity %v", l.ID, l.Quantity), true + } + return "", false } // usageLiner is any domain (or composite part) that can project its priced @@ -347,6 +470,7 @@ func wantedKinds(sel []string) (map[domain.Kind]bool, error) { domain.ECSFargate: true, domain.Lambda: true, domain.K8sFargate: true, + domain.RDS: true, } if len(sel) == 0 { return buildable, nil @@ -532,14 +656,41 @@ func runDomainsReport(w io.Writer, args []string) error { // somebody might put in a business case. return fmt.Errorf("aggregate report failed validation (this is a bug): %w", err) } + // --rds-detail adds pkg/rds's own report. It is a SECOND rendering of the + // same findings, and it earns its place because the layouts disagree on + // purpose: the aggregate leads with money and lists refusals under it, + // while pkg/rds leads with the refusals and puts the money second. In this + // domain the refusal IS the finding, and a reader who sees a dollar figure + // first reads the refusal as a caveat on a recommendation that does not + // exist. + var rdsReport *krds.Report + if df.rdsDetail && rt.rds != nil { + rdsReport = rt.rds.Report(rt.Now, rt.Ledger) + if rdsReport != nil { + if err := rdsReport.Validate(); err != nil { + return fmt.Errorf("rds report failed validation (this is a bug): %w", err) + } + } + } + if df.jsonOut { - return writeJSON(w, map[string]any{ - "report": rep, "warnings": rt.Warnings, - }) + out := map[string]any{"report": rep, "warnings": rt.Warnings} + if rdsReport != nil { + out["rds"] = rdsReport + } + return writeJSON(w, out) } if err := rep.WriteText(w); err != nil { return err } + if rdsReport != nil { + if _, err := io.WriteString(w, "\n"); err != nil { + return err + } + if err := rdsReport.WriteText(w); err != nil { + return err + } + } var b strings.Builder writeWarnings(&b, rt.Warnings) _, err = io.WriteString(w, b.String()) diff --git a/cmd/kilter/domains_test.go b/cmd/kilter/domains_test.go index 630ed7a..77e21fc 100644 --- a/cmd/kilter/domains_test.go +++ b/cmd/kilter/domains_test.go @@ -348,7 +348,7 @@ func TestNoDomainCanPlanAStepInThisBuild(t *testing.T) { Plans []domain.Plan `json:"plans"` } runJSON(t, &env, baseArgs(t, "plan")...) - if len(env.Plans) != 4 { + if len(env.Plans) != 5 { t.Fatalf("got %d plans, want one per wired domain", len(env.Plans)) } for _, p := range env.Plans { @@ -464,7 +464,7 @@ func TestBadInputIsRefusedRatherThanGuessed(t *testing.T) { }{ {"unknown subcommand", []string{"frobnicate"}, "unknown subcommand"}, {"no subcommand", nil, "subcommand is required"}, - {"unknown domain", []string{"report", "--domain", "rds"}, "unknown domain"}, + {"unknown domain", []string{"report", "--domain", "quantum-annealer"}, "unknown domain"}, {"known but unwired domain", []string{"report", "--domain", "k8s-nodes"}, "not wired into this binary"}, {"missing snapshot file", []string{"report", "--snapshot", "testdata/nope.json"}, "no such file"}, {"bad --now", []string{"report", "--now", "yesterday"}, "--now"}, diff --git a/cmd/kilter/explain.go b/cmd/kilter/explain.go new file mode 100644 index 0000000..9a42d57 --- /dev/null +++ b/cmd/kilter/explain.go @@ -0,0 +1,609 @@ +package main + +import ( + "encoding/json" + "flag" + "fmt" + "io" + "os" + "sort" + "strings" + "time" + + "github.com/agenticode/kilter/pkg/actuate" + "github.com/agenticode/kilter/pkg/api" + "github.com/agenticode/kilter/pkg/evidence" + "github.com/agenticode/kilter/pkg/explain" + "github.com/agenticode/kilter/pkg/model" + "github.com/agenticode/kilter/pkg/plan" + "github.com/agenticode/kilter/pkg/pricing" + "github.com/agenticode/kilter/pkg/recommend" +) + +// `kilter explain` and `kilter why-cost` — the explanation plane, made +// runnable. +// +// Both commands read a SEQUENCE of recorded cluster snapshots, the format +// `kilter analyze --dump-snapshot` already writes. That is the honest input: +// pkg/explain has no clock and no network, its window is an argument, and both +// entry points need history that pkg/store does not keep (see +// cmd/WIRING-FINDINGS.md). Feeding them recorded snapshots is the same +// discipline `kilter domains` and `kilter simulate` already use. +// +// # The publish gate is not optional +// +// pkg/explain's central rule is §5.6/§5.7's: a number the reader cannot trace +// is worse than a missing one, so every Term, every Driver and the residual +// carry an evidence ID, and every ID must resolve against the same store that +// produced the answer. Nothing inside pkg/explain enforces that at serve time +// — Verify exists and someone has to call it. These commands call it before +// printing, and treat a failure as an error rather than rendering an answer +// with a dangling citation. + +const explainUsage = `kilter explain — why the engine would resize this container + +Usage: + kilter explain --kube-snapshot PATH [--kube-snapshot PATH ...] \ + --workload Kind/namespace/name --container NAME [flags] + +Snapshots are replayed in timestamp order through the real recommender, so at +least two are needed before anything can be said. Nothing here calls a cluster. + +Flags: + --kube-snapshot PATH cluster snapshot JSON (repeatable, any order) + --workload REF Kind/namespace/name, e.g. Deployment/default/api + --container NAME container within the workload + --from RFC3339 evidence window start (default: the first snapshot) + --to RFC3339 evidence window end (default: the last snapshot) + --catalog PATH pricing catalog JSON (default: embedded) + --json emit the payload instead of the prose +` + +const whyCostUsage = `kilter why-cost — an additive, individually-citable cost decomposition + +Usage: + kilter why-cost --kube-snapshot PATH [--kube-snapshot PATH ...] \ + --from RFC3339 --to RFC3339 [flags] + +--from and --to are REQUIRED. There is no default window: the window is an +argument, and a wall-clock default makes a stored answer unreplayable. + +The decomposition is over the HOURLY RUN RATE, not integrated spend, and it +satisfies sum(terms) + residual == delta exactly — every amount is an int64 +count of millionths of a dollar, so the sum is order-independent by +construction. + +Flags: + --kube-snapshot PATH cluster snapshot JSON (repeatable, any order) + --from RFC3339 window start, inclusive (required) + --to RFC3339 window end, EXCLUSIVE (required) — [from, to), matching + pkg/evidence's window convention + --ledger PATH kilter ledger --json output, for the kilter-action term + --catalog PATH pricing catalog JSON (default: embedded) + --json emit the attribution instead of the prose +` + +func runExplain(args []string) error { return runExplainTo(os.Stdout, args) } +func runWhyCost(args []string) error { return runWhyCostTo(os.Stdout, args) } + +// ---------------------------------------------------------------- why-cost + +func runWhyCostTo(w io.Writer, args []string) error { + fs := flag.NewFlagSet("why-cost", flag.ContinueOnError) + fs.SetOutput(w) + var snaps repeatedFlag + fs.Var(&snaps, "kube-snapshot", "cluster snapshot JSON (repeatable)") + from := fs.String("from", "", "window start (RFC3339, required)") + to := fs.String("to", "", "window end (RFC3339, required)") + ledgerPath := fs.String("ledger", "", "kilter ledger --json output") + catalogPath := fs.String("catalog", "", "pricing catalog JSON") + jsonOut := fs.Bool("json", false, "emit JSON") + if err := fs.Parse(args); err != nil { + return err + } + if len(snaps) == 0 { + fmt.Fprint(w, whyCostUsage) + return fmt.Errorf("why-cost: at least one --kube-snapshot is required") + } + if *from == "" || *to == "" { + fmt.Fprint(w, whyCostUsage) + return fmt.Errorf("why-cost: --from and --to are required; the window is an argument, " + + "and an answer computed over a wall-clock default cannot be replayed") + } + start, err := time.Parse(time.RFC3339, *from) + if err != nil { + return fmt.Errorf("--from: %w", err) + } + end, err := time.Parse(time.RFC3339, *to) + if err != nil { + return fmt.Errorf("--to: %w", err) + } + if !end.After(start) { + return fmt.Errorf("why-cost: --to must be after --from") + } + catalog, err := loadCatalog(*catalogPath) + if err != nil { + return err + } + series, err := loadSnapshotSeries(snaps) + if err != nil { + return err + } + cluster := series[0].ClusterID + + store, err := evidence.NewMemory(evidence.Config{}) + if err != nil { + return err + } + // Observe a timeline point per snapshot in the window. The point's cost is + // pkg/pricing's own SnapshotCost, which includes Fargate; the composition + // below excludes Fargate nodes, so the difference lands in the residual + // with a note. That gap is honest and coarse, and naming it beats hiding + // it inside a term. + // [from, to) — HALF-OPEN, matching pkg/evidence's own window convention + // and the filter explain.WhyCost applies internally. Observing on a + // different convention than the consumer filters on would record a point + // the decomposition then ignores, and the count check below would pass + // while WhyCost failed. + var inWindow []*model.ClusterSnapshot + for _, s := range series { + if s.Timestamp.Before(start) || !s.Timestamp.Before(end) { + continue + } + cost := catalog.SnapshotCost(s) + if err := store.ObservePoint(cluster, evidence.TimelinePoint{ + At: s.Timestamp, CostUSDPerHour: cost.HourlyUSD, Nodes: countPricedNodes(s), + }); err != nil { + return fmt.Errorf("why-cost: record timeline: %w", err) + } + inWindow = append(inWindow, s) + } + observed := len(inWindow) + if observed < 2 { + return fmt.Errorf("why-cost: %d snapshot(s) inside [%s, %s) — a change needs two "+ + "observations, and one timeline point is not a change. Note the window is "+ + "half-open: a snapshot at exactly --to is outside it", + observed, start.UTC().Format(time.RFC3339), end.UTC().Format(time.RFC3339)) + } + timeline, err := store.Timeline(cluster, start, end) + if err != nil { + return err + } + + actions, err := loadLedgerActions(*ledgerPath, cluster, start, end) + if err != nil { + return err + } + + // The two edges are the FIRST and LAST snapshots that produced a timeline + // point in the window, not the newest snapshot on either side of the + // requested boundary. They must be the same two instants the measured + // ΔCost was taken between: a composition describing t+24h against a + // measurement taken at t+12h would push a real, explainable fleet change + // into the residual and call it unexplained. + att, err := explain.WhyCost(explain.Input{ + Cluster: cluster, From: start, To: end, + Timeline: timeline, + Start: basisFrom(inWindow[0], catalog), + End: basisFrom(inWindow[observed-1], catalog), + Actions: actions, + }) + if err != nil { + // A decomposition that cannot be computed is an error, not an empty + // table. + return fmt.Errorf("why-cost: %w", err) + } + // §5.7's publish gate. Nothing else enforces it. + if err := att.Verify(explain.Resolver{Store: store, Actions: actions}); err != nil { + return fmt.Errorf("why-cost: the answer has a citation that does not resolve, so it is not "+ + "publishable: %w", err) + } + if *jsonOut { + return writeJSON(w, att) + } + _, err = io.WriteString(w, att.Prose()+"\n") + return err +} + +// basisFrom prices one snapshot's fleet composition. +// +// Three things pkg/explain/FINDINGS.md says the wiring must get right, all +// here: +// +// 1. FARGATE NODES ARE EXCLUDED. A Fargate "node" is a single-pod VM billed +// per quantized pod, not a shareable machine; pricing it per node shape +// would inflate the fleet total and put the error inside a term instead of +// in the residual. +// 2. An empty fleet is &CostBasis{At: t}, NOT nil. nil means "I could not +// determine the composition"; a cluster created inside the window really +// was empty at the start edge, and conflating the two silently downgrades +// a complete answer into a residual. +// 3. Namespace demand is REQUESTED capacity, not usage — requests are what +// force node count. The clamping rule is pkg/plan's clampedRequests +// (negatives to zero), duplicated rather than re-invented because that +// function is unexported and pkg/plan is not this unit's to change. +func basisFrom(snap *model.ClusterSnapshot, cat *pricing.Catalog) *explain.CostBasis { + if snap == nil { + return nil + } + b := &explain.CostBasis{At: snap.Timestamp} + + type groupKey struct { + instanceType string + spot bool + } + groups := map[groupKey]*explain.NodeGroup{} + for i := range snap.Nodes { + n := &snap.Nodes[i] + if n.IsFargate() { + continue + } + hourly, _ := cat.NodeHourlyCost(n) + k := groupKey{n.InstanceType, n.Spot} + g := groups[k] + if g == nil { + g = &explain.NodeGroup{InstanceType: n.InstanceType, Spot: n.Spot, UnitUSDPerHour: hourly} + groups[k] = g + } + g.Nodes++ + } + keys := make([]groupKey, 0, len(groups)) + for k := range groups { + keys = append(keys, k) + } + // Sorted, because a map range is not an order and this slice is summed. + sort.Slice(keys, func(i, j int) bool { + if keys[i].instanceType != keys[j].instanceType { + return keys[i].instanceType < keys[j].instanceType + } + return !keys[i].spot && keys[j].spot + }) + for _, k := range keys { + b.Groups = append(b.Groups, *groups[k]) + } + + demand := map[string]*explain.NamespaceDemand{} + for i := range snap.Pods { + p := &snap.Pods[i] + if p.Phase == "Succeeded" || p.Phase == "Failed" { + continue + } + req := clampedPodRequests(p) + d := demand[p.Namespace] + if d == nil { + d = &explain.NamespaceDemand{Namespace: p.Namespace} + demand[p.Namespace] = d + } + d.MilliCPU += req.MilliCPU + d.MemoryBytes += req.MemoryBytes + d.Pods++ + } + ns := make([]string, 0, len(demand)) + for n := range demand { + ns = append(ns, n) + } + sort.Strings(ns) + for _, n := range ns { + b.Namespaces = append(b.Namespaces, *demand[n]) + } + return b +} + +// clampedPodRequests mirrors pkg/plan's clampedRequests exactly: the pod's +// effective request with negatives clamped to zero. It is duplicated rather +// than re-derived — pkg/explain/FINDINGS.md is explicit that a third +// definition of "requested capacity" is the thing to avoid. +func clampedPodRequests(p *model.PodSpec) model.Resources { + req := p.Requests() + if req.MilliCPU < 0 { + req.MilliCPU = 0 + } + if req.MemoryBytes < 0 { + req.MemoryBytes = 0 + } + return req +} + +// countPricedNodes counts the nodes the composition prices, so the timeline's +// node count and the composition's agree about what a "node" is. +func countPricedNodes(snap *model.ClusterSnapshot) int { + var n int + for i := range snap.Nodes { + if !snap.Nodes[i].IsFargate() { + n++ + } + } + return n +} + +// ---------------------------------------------------------------- explain + +func runExplainTo(w io.Writer, args []string) error { + fs := flag.NewFlagSet("explain", flag.ContinueOnError) + fs.SetOutput(w) + var snaps repeatedFlag + fs.Var(&snaps, "kube-snapshot", "cluster snapshot JSON (repeatable)") + workload := fs.String("workload", "", "Kind/namespace/name") + container := fs.String("container", "", "container name") + from := fs.String("from", "", "evidence window start (RFC3339)") + to := fs.String("to", "", "evidence window end (RFC3339)") + catalogPath := fs.String("catalog", "", "pricing catalog JSON") + jsonOut := fs.Bool("json", false, "emit JSON") + if err := fs.Parse(args); err != nil { + return err + } + if len(snaps) == 0 || *workload == "" || *container == "" { + fmt.Fprint(w, explainUsage) + return fmt.Errorf("explain: --kube-snapshot, --workload and --container are required") + } + ref, err := parseWorkloadRef(*workload) + if err != nil { + return err + } + catalog, err := loadCatalog(*catalogPath) + if err != nil { + return err + } + series, err := loadSnapshotSeries(snaps) + if err != nil { + return err + } + cluster := series[0].ClusterID + key := model.ContainerKey{Workload: ref, Container: *container} + + // Replay through the REAL recommender, in the same observe-then-ask order + // pkg/api's Ingest/Plan runs, and fill the evidence substrate from the + // same snapshots. The window is resolved to concrete timestamps BEFORE + // anything is computed and echoed in the output, so the answer is + // replayable. + // The default window is the span the EVIDENCE covers, not the span the + // snapshot timestamps cover: one snapshot can carry days of usage samples, + // and a window taken from the snapshot instants alone would be empty for a + // single-snapshot history while the substrate holds three days of it. + start, end := evidenceSpan(series) + if *from != "" { + if start, err = time.Parse(time.RFC3339, *from); err != nil { + return fmt.Errorf("--from: %w", err) + } + } + if *to != "" { + if end, err = time.Parse(time.RFC3339, *to); err != nil { + return fmt.Errorf("--to: %w", err) + } + } + if !end.After(start) { + return fmt.Errorf("explain: the evidence window [%s, %s] is empty", + start.UTC().Format(time.RFC3339), end.UTC().Format(time.RFC3339)) + } + + store, err := evidence.NewMemory(evidence.Config{}) + if err != nil { + return err + } + rec, err := recommend.New(recommend.DefaultConfig()) + if err != nil { + return err + } + for _, s := range series { + rec.ObserveSnapshot(s) + if err := observeUsage(store, cluster, s); err != nil { + return err + } + if err := store.ObservePoint(cluster, evidence.TimelinePoint{ + At: s.Timestamp, CostUSDPerHour: catalog.SnapshotCost(s).HourlyUSD, + Nodes: countPricedNodes(s), + }); err != nil { + return err + } + } + + var found *recommend.Recommendation + for _, r := range rec.Recommendations(series[len(series)-1]) { + if r.Key == key { + found = &r + break + } + } + + req := explain.ExplainRequest{ + Cluster: cluster, + Subject: evidence.ContainerSubject(cluster, key), + From: start, To: end, + Store: store, + Rec: found, + } + payload, err := explain.BuildExplain(req) + if err != nil { + return fmt.Errorf("explain: %w", err) + } + // §5.7's publish gate, before anything is shown. + if err := payload.Verify(explain.Resolver{Store: store}); err != nil { + return fmt.Errorf("explain: the answer has a citation that does not resolve, so it is not "+ + "publishable: %w", err) + } + if *jsonOut { + return writeJSON(w, payload) + } + var b strings.Builder + fmt.Fprintf(&b, "kilter explain — %s over [%s, %s]\n\n", key.String(), + start.UTC().Format(time.RFC3339), end.UTC().Format(time.RFC3339)) + b.WriteString(payload.Prose()) + b.WriteString("\n") + _, err = io.WriteString(w, b.String()) + return err +} + +// observeUsage feeds one snapshot's measured container usage into the +// substrate, which is what BuildExplain reads its digests and series from. +func observeUsage(store *evidence.Memory, cluster string, snap *model.ClusterSnapshot) error { + for i := range snap.Usage { + u := &snap.Usage[i] + at := u.Timestamp + if at.IsZero() { + at = snap.Timestamp + } + // model.Usage carries CPU, memory and a window and nothing else: no + // throttle ratio, no restart or OOM delta. Those reach the substrate + // through evidence events from other collectors, so they are left + // zero here rather than fabricated — "signal absent" is a state + // pkg/decision already knows how to read, and a zero throttle ratio + // invented by the wiring would be a claim nobody measured. + if err := store.ObserveSample(evidence.ContainerSubject(cluster, u.Key), evidence.Sample{ + At: at, MilliCPU: u.MilliCPU, MemoryBytes: u.MemoryBytes, + }); err != nil { + return fmt.Errorf("explain: record usage: %w", err) + } + } + return nil +} + +// ---------------------------------------------------------------- ledger + +// ledgerFile is the shape `kilter ledger --json` writes. +type ledgerFile struct { + Entries []api.LedgerEntry `json:"entries"` +} + +// loadLedgerActions projects Kilter's own audit ledger into the shape the +// attribution needs, filtered to this cluster and window. +// +// This is pkg/explain/FINDINGS.md §1's mapping, field for field, and the two +// things it insists on are both here: +// +// - Applied must be EXACT. A dry-run moved no money, so counting one would +// attribute a cost change to a plan that changed nothing — which is the +// classic attribution lie with a plan attached. Only StatusDone counts; +// StatusDryRun is deliberately excluded even though actuate.Report counts +// it as done for its own purposes. +// - NodesAdded stays 0 because no plan type provisions a node today. A +// field that could only ever be wrong is left at zero rather than guessed. +// +// Finished has no ledger field, so it is left zero and pkg/explain falls back +// to At (TestZeroFinishedTimeFallsBackToStart). +func loadLedgerActions(path, cluster string, from, to time.Time) ([]explain.LedgerAction, error) { + if path == "" { + return nil, nil + } + raw, err := os.ReadFile(path) + if err != nil { + return nil, fmt.Errorf("--ledger: %w", err) + } + var lf ledgerFile + if err := json.Unmarshal(raw, &lf); err != nil { + return nil, fmt.Errorf("%s: %w", path, err) + } + out := make([]explain.LedgerAction, 0, len(lf.Entries)) + for _, e := range lf.Entries { + c := e.Cluster + if c == "" { + c = cluster + } + if c != cluster || e.At.Before(from) || e.At.After(to) { + continue + } + a := explain.LedgerAction{ + At: e.At, Cluster: c, Fingerprint: e.Fingerprint, + Mode: e.Mode, Risk: e.Risk, + Applied: e.Mode == "apply" && e.Done > 0, + CostBeforeHourlyUSD: e.CostBeforeHourlyUSD, + ProjectedHourlyUSD: e.ProjectedHourlyUSD, + } + for _, s := range e.Steps { + if s.Status != actuate.StatusDone { + continue + } + switch s.Step.Type { + case plan.StepDeleteNode: + a.NodesRemoved++ + case plan.StepResizeWorkload: + a.Resizes++ + } + } + out = append(out, a) + } + // Sorted: a ledger read in file order would make the attribution depend on + // the order entries happened to be appended. + sort.SliceStable(out, func(i, j int) bool { + if !out[i].At.Equal(out[j].At) { + return out[i].At.Before(out[j].At) + } + return out[i].Fingerprint < out[j].Fingerprint + }) + return out, nil +} + +// ---------------------------------------------------------------- shared + +// loadSnapshotSeries reads every snapshot and returns them in timestamp order. +// +// Order is imposed here rather than trusted from the command line: a replay +// whose result depends on the order paths were typed in is not a function of +// its inputs. Two snapshots sharing a timestamp are rejected for the same +// reason pkg/backtest rejects them — a tie has no defined replay order. +func loadSnapshotSeries(paths []string) ([]*model.ClusterSnapshot, error) { + if len(paths) == 0 { + return nil, fmt.Errorf("no snapshot supplied") + } + out := make([]*model.ClusterSnapshot, 0, len(paths)) + for _, p := range paths { + raw, err := os.ReadFile(p) + if err != nil { + return nil, fmt.Errorf("--kube-snapshot: %w", err) + } + var snap model.ClusterSnapshot + if err := json.Unmarshal(raw, &snap); err != nil { + return nil, fmt.Errorf("%s: %w", p, err) + } + out = append(out, &snap) + } + sort.SliceStable(out, func(i, j int) bool { return out[i].Timestamp.Before(out[j].Timestamp) }) + for i := 1; i < len(out); i++ { + if out[i].Timestamp.Equal(out[i-1].Timestamp) { + return nil, fmt.Errorf("two snapshots share the timestamp %s; a tie has no defined replay order", + out[i].Timestamp.UTC().Format(time.RFC3339)) + } + if out[i].ClusterID != out[0].ClusterID { + return nil, fmt.Errorf("snapshots name different clusters (%q and %q)", + out[0].ClusterID, out[i].ClusterID) + } + } + return out, nil +} + +// evidenceSpan is the closed interval every observation in the series falls +// inside — snapshot instants and usage sample timestamps alike. +func evidenceSpan(series []*model.ClusterSnapshot) (time.Time, time.Time) { + start, end := series[0].Timestamp, series[len(series)-1].Timestamp + for _, s := range series { + if s.Timestamp.Before(start) { + start = s.Timestamp + } + if s.Timestamp.After(end) { + end = s.Timestamp + } + for i := range s.Usage { + at := s.Usage[i].Timestamp + if at.IsZero() { + continue + } + if at.Before(start) { + start = at + } + if at.After(end) { + end = at + } + } + } + // The window is half-open in pkg/evidence, so the last observation must + // fall strictly inside it. + return start, end.Add(time.Second) +} + +// parseWorkloadRef parses Kind/namespace/name. +func parseWorkloadRef(s string) (model.WorkloadRef, error) { + parts := strings.Split(s, "/") + if len(parts) != 3 || parts[0] == "" || parts[1] == "" || parts[2] == "" { + return model.WorkloadRef{}, fmt.Errorf("--workload %q: want Kind/namespace/name, e.g. Deployment/default/api", s) + } + return model.WorkloadRef{Kind: model.WorkloadKind(parts[0]), Namespace: parts[1], Name: parts[2]}, nil +} diff --git a/cmd/kilter/explain_test.go b/cmd/kilter/explain_test.go new file mode 100644 index 0000000..d878639 --- /dev/null +++ b/cmd/kilter/explain_test.go @@ -0,0 +1,479 @@ +package main + +import ( + "encoding/json" + "io" + "os" + "path/filepath" + "strings" + "testing" + "time" + + "github.com/agenticode/kilter/pkg/actuate" + "github.com/agenticode/kilter/pkg/api" + "github.com/agenticode/kilter/pkg/explain" + "github.com/agenticode/kilter/pkg/model" + "github.com/agenticode/kilter/pkg/plan" +) + +// These tests drive pkg/explain's REAL decomposition and REAL explain payload +// through the REAL CLI entry point, over recorded cluster snapshots. Both +// commands call Verify before printing, so a citation that does not resolve is +// a test failure rather than a rendered answer. + +var whyCostT0 = time.Date(2026, 8, 20, 0, 0, 0, 0, time.UTC) + +// fleetSnapshot builds one priced fleet at an instant: n on-demand m5.large +// plus s spot m5.large, with one namespace's requested capacity. +func fleetSnapshot(at time.Time, onDemand, spot int, nsMilliCPU int64) *model.ClusterSnapshot { + snap := &model.ClusterSnapshot{ClusterID: "why-cost-demo", Timestamp: at} + add := func(n int, isSpot bool, prefix string) { + for i := 0; i < n; i++ { + snap.Nodes = append(snap.Nodes, model.NodeSpec{ + Name: prefix + string(rune('a'+i)), + Capacity: model.Resources{MilliCPU: 2000, MemoryBytes: 8 << 30}, + Allocatable: model.Resources{MilliCPU: 1900, MemoryBytes: 7 << 30}, + Ready: true, InstanceType: "m5.large", Spot: isSpot, + Provider: "aws", Region: "us-east-1", Zone: "us-east-1a", + }) + } + } + add(onDemand, false, "od-") + add(spot, true, "spot-") + + ref := model.WorkloadRef{Kind: model.KindDeployment, Namespace: "shop", Name: "web"} + snap.Pods = append(snap.Pods, model.PodSpec{ + UID: "pod-shop", Name: "web-0", Namespace: "shop", Workload: ref, + NodeName: "od-a", Phase: "Running", + Containers: []model.ContainerSpec{{ + Name: "web", + Requests: model.Resources{MilliCPU: nsMilliCPU, MemoryBytes: 1 << 30}, + }}, + }) + snap.Workloads = append(snap.Workloads, model.WorkloadInfo{Ref: ref, Replicas: 1, Ready: 1}) + return snap +} + +func writeSnapshot(t *testing.T, dir, name string, snap *model.ClusterSnapshot) string { + t.Helper() + raw, err := json.Marshal(snap) + if err != nil { + t.Fatal(err) + } + path := filepath.Join(dir, name) + if err := os.WriteFile(path, raw, 0o644); err != nil { + t.Fatal(err) + } + return path +} + +// whyCostFleet writes a three-point history in which the fleet grows from four +// on-demand nodes to six nodes, two of which are spot: node count moves AND +// capacity type moves, which is exactly the overlap the attribution order +// exists to resolve. +func whyCostFleet(t *testing.T) (dir string, args []string) { + t.Helper() + dir = t.TempDir() + var out []string + for i, s := range []*model.ClusterSnapshot{ + fleetSnapshot(whyCostT0, 4, 0, 500), + fleetSnapshot(whyCostT0.Add(12*time.Hour), 5, 1, 700), + fleetSnapshot(whyCostT0.Add(24*time.Hour), 4, 2, 900), + } { + out = append(out, "--kube-snapshot", + writeSnapshot(t, dir, "snap-"+string(rune('0'+i))+".json", s)) + } + return dir, out +} + +func runWhyCostOK(t *testing.T, args ...string) string { + t.Helper() + var b strings.Builder + if err := runWhyCostTo(&b, args); err != nil { + t.Fatalf("kilter why-cost: %v\n%s", err, b.String()) + } + return b.String() +} + +func whyCostAttribution(t *testing.T, args ...string) *explain.Attribution { + t.Helper() + raw := runWhyCostOK(t, append(args, "--json")...) + var att explain.Attribution + if err := json.Unmarshal([]byte(raw), &att); err != nil { + t.Fatalf("decode attribution: %v\n%s", err, raw) + } + return &att +} + +// TestWhyCostIsReachableAndAdditive. +// +// The invariant pkg/explain exists to protect, asserted at the CLI: +// sum(terms) + residual == delta, EXACTLY. Every amount is an int64 count of +// µUSD, so this is integer arithmetic and "exactly" is meant literally. +func TestWhyCostIsReachableAndAdditive(t *testing.T) { + _, snaps := whyCostFleet(t) + att := whyCostAttribution(t, append(snaps, + "--from", whyCostT0.Format(time.RFC3339), + "--to", whyCostT0.Add(24*time.Hour).Format(time.RFC3339))...) + + var sum explain.Micro + for _, term := range att.Terms { + sum += term.Micro + } + if sum+att.Residual.Micro != att.DeltaMicro { + t.Errorf("sum(terms)=%d + residual=%d != delta=%d", + sum, att.Residual.Micro, att.DeltaMicro) + } + if len(att.Residual.Evidence) == 0 { + t.Error("the residual ships uncited") + } + if len(att.Terms) == 0 { + t.Fatal("no terms; the decomposition explained nothing") + } + // Every term, sub-term and the residual must carry a citation. A number + // the reader cannot trace is worse than a missing one. + for _, term := range att.Terms { + if len(term.Evidence) == 0 { + t.Errorf("term %q ships uncited", term.Kind) + } + var subs explain.Micro + for _, sub := range term.Of { + subs += sub.Micro + if len(sub.Evidence) == 0 { + t.Errorf("sub-term %q of %q ships uncited", sub.Kind, term.Kind) + } + } + if len(term.Of) > 0 && subs != term.Micro { + t.Errorf("sum(%q.Of)=%d != %d", term.Kind, subs, term.Micro) + } + } + if len(att.Order) == 0 { + t.Error("the attribution does not state the convention it was computed under") + } +} + +// TestWhyCostNamesTheFactorsThatActuallyMoved. The fleet went from 4 on-demand +// nodes to 4 on-demand + 2 spot, so node count and spot ratio must both carry +// a non-zero term — otherwise the decomposition is arithmetically valid and +// tells the operator nothing. +func TestWhyCostNamesTheFactorsThatActuallyMoved(t *testing.T) { + _, snaps := whyCostFleet(t) + att := whyCostAttribution(t, append(snaps, + "--from", whyCostT0.Format(time.RFC3339), + "--to", whyCostT0.Add(24*time.Hour).Format(time.RFC3339))...) + + byKind := map[string]explain.Micro{} + for _, term := range att.Terms { + byKind[string(term.Kind)] = term.Micro + } + if byKind["node-count"] == 0 { + t.Errorf("node count moved from 4 to 6 and the term is zero: %v", byKind) + } + if byKind["spot-ratio"] >= 0 { + t.Errorf("two nodes moved to spot and the term is %d; want a saving", byKind["spot-ratio"]) + } + if att.DeltaMicro == 0 { + t.Error("the measured delta is zero; the fixture proves nothing") + } + // And the prose renders it rather than only the JSON. + out := runWhyCostOK(t, append(snaps, + "--from", whyCostT0.Format(time.RFC3339), + "--to", whyCostT0.Add(24*time.Hour).Format(time.RFC3339))...) + for _, want := range []string{"node-count", "spot-ratio"} { + if !strings.Contains(out, want) { + t.Errorf("the prose does not mention %q:\n%s", want, out) + } + } +} + +// TestWhyCostRequiresItsWindow. +// +// pkg/explain has no clock on purpose: the window is an argument, so the same +// inputs give the same answer forever. A wall-clock default would make every +// stored answer unreplayable, so the CLI refuses rather than inventing one. +func TestWhyCostRequiresItsWindow(t *testing.T) { + _, snaps := whyCostFleet(t) + for _, tc := range []struct { + name, want string + args []string + }{ + {"no window", "--from and --to are required", snaps}, + {"only from", "--from and --to are required", + append(append([]string{}, snaps...), "--from", whyCostT0.Format(time.RFC3339))}, + {"backwards", "--to must be after --from", + append(append([]string{}, snaps...), + "--from", whyCostT0.Add(24*time.Hour).Format(time.RFC3339), + "--to", whyCostT0.Format(time.RFC3339))}, + {"one observation", "one timeline point is not a change", + append(append([]string{}, snaps...), + "--from", whyCostT0.Format(time.RFC3339), + "--to", whyCostT0.Add(time.Hour).Format(time.RFC3339))}, + } { + t.Run(tc.name, func(t *testing.T) { + var b strings.Builder + err := runWhyCostTo(&b, tc.args) + if err == nil { + t.Fatalf("accepted:\n%s", b.String()) + } + if !strings.Contains(err.Error(), tc.want) { + t.Errorf("error = %q, want it to mention %q", err, tc.want) + } + }) + } +} + +// TestWhyCostIsOrderAndRepeatIndependent: the answer must not depend on the +// order the snapshots were typed on the command line, nor on Go's map +// iteration order within a process. +func TestWhyCostIsOrderAndRepeatIndependent(t *testing.T) { + dir, snaps := whyCostFleet(t) + window := []string{ + "--from", whyCostT0.Format(time.RFC3339), + "--to", whyCostT0.Add(24 * time.Hour).Format(time.RFC3339), + } + base := runWhyCostOK(t, append(append([]string{}, snaps...), window...)...) + for i := 0; i < 6; i++ { + if got := runWhyCostOK(t, append(append([]string{}, snaps...), window...)...); got != base { + t.Fatalf("repeat %d differs in the same process", i) + } + } + reversed := []string{ + "--kube-snapshot", filepath.Join(dir, "snap-2.json"), + "--kube-snapshot", filepath.Join(dir, "snap-0.json"), + "--kube-snapshot", filepath.Join(dir, "snap-1.json"), + } + if got := runWhyCostOK(t, append(reversed, window...)...); got != base { + t.Error("the answer depends on the order snapshots were supplied in") + } +} + +// TestFargateIsExcludedFromTheCompositionAndLandsInTheResidual. +// +// A Fargate "node" is a single-pod VM billed per quantized pod, not a +// shareable machine, so pricing it per node shape would inflate the fleet +// total and put the error inside a TERM. Excluding it moves the gap to the +// residual, where it is visible — the correct behaviour, and a worse answer +// than pricing Fargate separately, which is why it is reported rather than +// absorbed. +func TestFargateIsExcludedFromTheCompositionAndLandsInTheResidual(t *testing.T) { + dir := t.TempDir() + withFargate := fleetSnapshot(whyCostT0, 4, 0, 500) + withFargate.Nodes = append(withFargate.Nodes, model.NodeSpec{ + Name: "fargate-1", ManagedBy: model.ManagedByFargate, + Labels: map[string]string{model.LabelComputeType: "fargate"}, + Capacity: model.Resources{MilliCPU: 96000, MemoryBytes: 384 << 30}, + Allocatable: model.Resources{MilliCPU: 96000, MemoryBytes: 384 << 30}, + Ready: true, Provider: "aws", Region: "us-east-1", Zone: "us-east-1a", + }) + withFargate.Pods = append(withFargate.Pods, model.PodSpec{ + UID: "pod-fg", Name: "job-0", Namespace: "batch", NodeName: "fargate-1", + Workload: model.WorkloadRef{Kind: model.KindDeployment, Namespace: "batch", Name: "job"}, + Phase: "Running", + Containers: []model.ContainerSpec{{ + Name: "job", Requests: model.Resources{MilliCPU: 1000, MemoryBytes: 2 << 30}, + }}, + ProvisionedCapacity: model.Resources{MilliCPU: 1000, MemoryBytes: 2 * 1000 * 1000 * 1000}, + }) + a := writeSnapshot(t, dir, "a.json", withFargate) + b := writeSnapshot(t, dir, "b.json", fleetSnapshot(whyCostT0.Add(24*time.Hour), 4, 0, 500)) + + // The window is half-open, so --to must be strictly after the last + // observation for that observation to be inside it. + att := whyCostAttribution(t, + "--kube-snapshot", a, "--kube-snapshot", b, + "--from", whyCostT0.Format(time.RFC3339), + "--to", whyCostT0.Add(25*time.Hour).Format(time.RFC3339)) + + // The fleet of m5.large nodes did not move, so every priced term is zero + // and the entire Fargate pod's cost is the residual — reported, never + // absorbed into the biggest term. + if att.Residual.Micro == 0 { + t.Fatalf("the Fargate cost was absorbed into a term rather than reported: %+v", att.Terms) + } + for _, term := range att.Terms { + if term.Kind == "node-count" && term.Micro != 0 { + t.Errorf("node count did not move and the term is %d", term.Micro) + } + } + if len(att.Notes) == 0 { + t.Error("the residual is unexplained and unremarked") + } +} + +// TestWhyCostAttributesAppliedActionsAndIgnoresDryRuns. +// +// A dry-run moved no money. Counting one would attribute a cost change to a +// plan that changed nothing — the classic attribution lie with a plan +// attached — so the projection sets Applied only for an applied entry with +// confirmed steps, and only StatusDone steps are counted. +func TestWhyCostAttributesAppliedActionsAndIgnoresDryRuns(t *testing.T) { + dir, snaps := whyCostFleet(t) + del := func(node string) actuate.StepStatus { + return actuate.StepStatus{ + Step: plan.Step{Type: plan.StepDeleteNode, Node: node}, + Status: actuate.StatusDone, + } + } + report := api.LedgerReport{Entries: []api.LedgerEntry{ + { + At: whyCostT0.Add(6 * time.Hour), Cluster: "why-cost-demo", Mode: "apply", + Fingerprint: "aaaa1111", Risk: "low", Done: 1, + Steps: []actuate.StepStatus{del("od-d")}, + }, + { + // A preview. It moved nothing and must be attributed nothing. + At: whyCostT0.Add(8 * time.Hour), Cluster: "why-cost-demo", Mode: "dry-run", + Fingerprint: "bbbb2222", Risk: "low", Done: 1, + Steps: []actuate.StepStatus{ + {Step: plan.Step{Type: plan.StepDeleteNode, Node: "od-c"}, Status: actuate.StatusDryRun}, + }, + }, + { + // Outside the window. + At: whyCostT0.Add(-48 * time.Hour), Cluster: "why-cost-demo", Mode: "apply", + Fingerprint: "cccc3333", Done: 1, Steps: []actuate.StepStatus{del("od-z")}, + }, + }} + raw, err := json.Marshal(report) + if err != nil { + t.Fatal(err) + } + ledgerPath := filepath.Join(dir, "ledger.json") + if err := os.WriteFile(ledgerPath, raw, 0o644); err != nil { + t.Fatal(err) + } + + actions, err := loadLedgerActions(ledgerPath, "why-cost-demo", + whyCostT0, whyCostT0.Add(24*time.Hour)) + if err != nil { + t.Fatal(err) + } + if len(actions) != 2 { + t.Fatalf("got %d actions in the window, want 2 (the third is outside it)", len(actions)) + } + applied, dryRun := actions[0], actions[1] + if !applied.Applied || applied.NodesRemoved != 1 { + t.Errorf("the applied entry projected to %+v", applied) + } + if dryRun.Applied { + t.Error("a dry-run was marked Applied; it moved no money") + } + if dryRun.NodesRemoved != 0 { + t.Errorf("a dry-run step was counted as a node removal (%d)", dryRun.NodesRemoved) + } + if applied.NodesAdded != 0 || dryRun.NodesAdded != 0 { + t.Error("NodesAdded must stay 0 until a plan type provisions nodes") + } + + // End to end: the kilter-action sub-term appears under node-count. + att := whyCostAttribution(t, append(append([]string{}, snaps...), + "--ledger", ledgerPath, + "--from", whyCostT0.Format(time.RFC3339), + "--to", whyCostT0.Add(24*time.Hour).Format(time.RFC3339))...) + var found bool + for _, term := range att.Terms { + for _, sub := range term.Of { + if sub.Kind == "kilter-action" { + found = true + } + } + } + if !found { + t.Errorf("no kilter-action sub-attribution: %+v", att.Terms) + } +} + +// TestExplainIsReachableAndCited. +// +// Every driver must be grounded in an evidence ID that resolves against the +// same store that produced the answer — §5.7's publish gate, which the CLI +// calls because nothing inside pkg/explain does it at serve time. +func TestExplainIsReachableAndCited(t *testing.T) { + args := []string{ + "--kube-snapshot", readFixture(t, "cluster.json"), + "--workload", "Deployment/default/api", "--container", "api", + } + var b strings.Builder + if err := runExplainTo(&b, append(args, "--json")); err != nil { + t.Fatalf("kilter explain: %v\n%s", err, b.String()) + } + var payload explain.Explanation + if err := json.Unmarshal([]byte(b.String()), &payload); err != nil { + t.Fatalf("decode: %v\n%s", err, b.String()) + } + if len(payload.Drivers) == 0 { + t.Fatal("the explanation has no drivers") + } + for _, d := range payload.Drivers { + if len(d.Evidence) == 0 { + t.Errorf("driver %q ships ungrounded", d.Kind) + } + } + if len(payload.Citations) == 0 { + t.Error("the payload cites nothing") + } + // The prose form echoes the window it was computed over, so a stored + // answer states its own window rather than depending on when it was read. + out := run2(t, runExplainTo, args) + if !strings.Contains(out, "over [") || !strings.Contains(out, "Z]") { + t.Errorf("the prose does not echo the resolved window:\n%s", out) + } + if !strings.Contains(out, "usage-history") { + t.Errorf("the prose does not render the drivers:\n%s", out) + } +} + +// TestExplainRefusesBadInput. +func TestExplainRefusesBadInput(t *testing.T) { + for _, tc := range []struct { + name string + args []string + want string + }{ + {"no snapshot", []string{"--workload", "Deployment/a/b", "--container", "c"}, "required"}, + {"bad workload ref", []string{ + "--kube-snapshot", readFixture(t, "cluster.json"), + "--workload", "api", "--container", "api"}, "Kind/namespace/name"}, + {"missing file", []string{ + "--kube-snapshot", "testdata/nope.json", + "--workload", "Deployment/a/b", "--container", "c"}, "no such file"}, + } { + t.Run(tc.name, func(t *testing.T) { + var b strings.Builder + if err := runExplainTo(&b, tc.args); err == nil { + t.Fatalf("accepted:\n%s", b.String()) + } else if !strings.Contains(err.Error(), tc.want) { + t.Errorf("error = %q, want it to mention %q", err, tc.want) + } + }) + } +} + +// TestSnapshotSeriesRejectsTiesAndMixedClusters: a tie has no defined replay +// order, and two clusters in one window is a different question. +func TestSnapshotSeriesRejectsTiesAndMixedClusters(t *testing.T) { + dir := t.TempDir() + a := writeSnapshot(t, dir, "a.json", fleetSnapshot(whyCostT0, 2, 0, 100)) + b := writeSnapshot(t, dir, "b.json", fleetSnapshot(whyCostT0, 3, 0, 100)) + if _, err := loadSnapshotSeries([]string{a, b}); err == nil || + !strings.Contains(err.Error(), "share the timestamp") { + t.Errorf("err = %v, want a duplicate-timestamp error", err) + } + other := fleetSnapshot(whyCostT0.Add(time.Hour), 2, 0, 100) + other.ClusterID = "somewhere-else" + c := writeSnapshot(t, dir, "c.json", other) + if _, err := loadSnapshotSeries([]string{a, c}); err == nil || + !strings.Contains(err.Error(), "different clusters") { + t.Errorf("err = %v, want a mixed-cluster error", err) + } +} + +// run2 runs a writer-based command and fails the test on error. +func run2(t *testing.T, fn func(io.Writer, []string) error, args []string) string { + t.Helper() + var b strings.Builder + if err := fn(&b, args); err != nil { + t.Fatalf("command failed: %v\n%s", err, b.String()) + } + return b.String() +} diff --git a/cmd/kilter/main.go b/cmd/kilter/main.go index f584130..5d4473c 100644 --- a/cmd/kilter/main.go +++ b/cmd/kilter/main.go @@ -6,8 +6,11 @@ // kilter controller executes brain plans under the safety envelope // kilter plan fetch & print the current plan from a brain // kilter insights predictive findings (OOM risk, saturation, capacity) -// kilter domains compute domains: ec2/ebs, ecs-fargate, lambda, k8s-fargate +// kilter domains compute domains: ec2/ebs, ecs-fargate, lambda, k8s-fargate, rds // kilter simulate replay a recorded snapshot through the decision engine +// kilter backtest score a policy against history (falsifiability harness) +// kilter explain the full explain payload behind one recommendation +// kilter why-cost an additive decomposition of a cluster's cost change // kilter version build information package main @@ -35,12 +38,15 @@ Commands: controller Execute brain plans (dry-run by default) plan Fetch and print the current plan from a brain insights Predictive findings: OOM risk, saturation, capacity exhaustion - domains Run every compute domain (ec2/ebs, ecs-fargate, lambda, k8s-fargate) + domains Run every compute domain (ec2/ebs, ecs-fargate, lambda, k8s-fargate, rds) pricing Sync live cloud prices into a catalog (sync-aws) ledger Audit trail: executed plans + measured cost curve (verifiable savings) approve Approve a plan fingerprint for --require-approval controllers undo Revert the most recent applied plan (resizes + cordons) simulate Replay a snapshot file through the decision engine + backtest Score a policy against history: oracle gap, regret, safety, gate + explain Why the engine would resize a container, with citations + why-cost Additive, individually-citable decomposition of a cost change version Print version Run "kilter -h" for command flags. @@ -79,6 +85,12 @@ func main() { err = runUndo(args) case "simulate": err = runSimulate(args) + case "backtest": + err = runBacktest(args) + case "explain": + err = runExplain(args) + case "why-cost": + err = runWhyCost(args) case "version", "--version", "-v": fmt.Printf("kilter %s (%s)\n", version, commit) case "help", "-h", "--help": diff --git a/cmd/kilter/rds.go b/cmd/kilter/rds.go new file mode 100644 index 0000000..12c5068 --- /dev/null +++ b/cmd/kilter/rds.go @@ -0,0 +1,163 @@ +package main + +import ( + "context" + "encoding/json" + "fmt" + "os" + "strings" + "time" + + krds "github.com/agenticode/kilter/pkg/rds" +) + +// The RDS wiring, and the exact line where it stops. +// +// pkg/rds/FINDINGS.md §6 owes cmd/ four things: the domain kind (landed in +// pkg/domain), an SDK adapter over three read seams, the collection loop, and +// the rate override. Three of the four are here. The fourth — the adapter over +// `*rds.Client` and `*cloudwatch.Client` — is NOT, and cannot be in this +// build: `github.com/aws/aws-sdk-go-v2/service/rds` and `.../service/cloudwatch` +// are not in go.mod, and adding them is a go.mod/go.sum change this unit may +// not make. See cmd/WIRING-FINDINGS.md. +// +// What replaces it is not a stub. `rds.Fixture` implements all three seams +// with real pagination, real truncation and real empty-account behaviour, and +// it is exported for exactly this reason — "the seams are the contract, and a +// contract nobody outside the package can exercise is not a contract". So +// --rds-fixture drives the REAL collector: rds.NewCollector over the recorded +// account, rds.Collector.Collect, the real window clamp, the real GetMetricData +// batching and ID routing. Every line of pkg/rds/collect.go that a live +// credential would exercise is exercised here, and the only thing missing is +// the field copy between an SDK struct and a struct with the same field names. +// +// No credential is read, no ~/.aws is opened, and no network call is made on +// this path — the same guarantee `kilter domains` already gives. + +// rdsFixtureFile is the on-disk shape of a recorded RDS account. +// +// It is a cmd/-side projection of `rds.Fixture` rather than that type +// directly, and the mismatch that forced it is worth naming: `rds.Fixture` +// carries `error` fields (InstancesErr, TagsErr, …) and a `Calls` counter +// struct, none of which have a JSON representation. Decoding straight into it +// would give a file format with four fields that silently cannot be set and +// one that pretends to be input. This projection carries the DATA half only, +// with explicit json tags, and turns the two seam-absence cases into the +// booleans the IAM table in §6.2 actually describes. +type rdsFixtureFile struct { + // Instances is the recorded rds:DescribeDBInstances inventory. + Instances []krds.DBInstanceRecord `json:"instances,omitempty"` + // Clusters is the recorded rds:DescribeDBClusters inventory. It is read + // for one reason: to tell an Aurora cluster from a Multi-AZ DB cluster + // without inferring it from a member's engine string (§5.3). + Clusters []krds.DBClusterRecord `json:"clusters,omitempty"` + // Tags maps a DB instance ARN to its rds:ListTagsForResource answer. + Tags map[string]map[string]string `json:"tags,omitempty"` + // Metrics maps "/" to datapoints. + Metrics map[string][]krds.Point `json:"metrics,omitempty"` + // Reservations is the recorded rds:DescribeReservedDBInstances inventory. + Reservations []krds.ReservedDBInstanceRecord `json:"reservations,omitempty"` + // PageSize splits every paginated response; 0 means one page. + PageSize int `json:"pageSize,omitempty"` + // DropResults omits the first N results from every GetMetricData page, + // reproducing a TRUNCATED response. A missing result is "we were not + // told", never "the metric is empty", and this is the knob that proves an + // idle verdict cannot be manufactured out of silence. + DropResults int `json:"dropResults,omitempty"` + + // NoMetricsAPI models a caller holding rds:Describe* and NOT + // cloudwatch:GetMetricData. §6.2: nil ⇒ every instance refuses with + // no-metric-evidence. That is a complete report, not a failed one. + NoMetricsAPI bool `json:"noMetricsAPI,omitempty"` + // NoCommitmentAPI models a caller without + // rds:DescribeReservedDBInstances. §6.2: nil ⇒ net == gross, which + // under-claims and can never invent a saving. + NoCommitmentAPI bool `json:"noCommitmentAPI,omitempty"` +} + +// collectRDS runs the real collector over a recorded account and returns the +// native snapshot. +// +// The window is [now-span, now] and is then CLAMPED by the collector, because +// 1-minute CloudWatch datapoints live 15 days: a snapshot that claims a 30-day +// window and holds 15 days of data is a lie told by omission, and every +// downstream "insufficient window" gate reads the claim rather than the data. +// The clamp is why c.Window() is rendered and the request is not. +func collectRDS(ctx context.Context, path, scope, region string, now time.Time, span time.Duration) (*krds.Snapshot, []string, error) { + raw, err := os.ReadFile(path) + if err != nil { + return nil, nil, fmt.Errorf("--rds-fixture: %w", err) + } + var ff rdsFixtureFile + dec := json.NewDecoder(strings.NewReader(string(raw))) + dec.DisallowUnknownFields() + if err := dec.Decode(&ff); err != nil { + return nil, nil, fmt.Errorf("%s: %w", path, err) + } + + fx := &krds.Fixture{ + Instances: ff.Instances, + Clusters: ff.Clusters, + Tags: ff.Tags, + Metrics: ff.Metrics, + Reservations: ff.Reservations, + PageSize: ff.PageSize, + DropResults: ff.DropResults, + } + + cfg := krds.DefaultCollectorConfig(krds.Window{Start: now.Add(-span), End: now}) + cfg.Scope, cfg.Region = scope, region + + // The three seams. Two of them are optional and their absence is a + // DIFFERENT report rather than a failure — that is the whole reason + // pkg/rds declares them separately instead of as one client interface. + var metrics krds.MetricsAPI + var reserved krds.CommitmentAPI + if !ff.NoMetricsAPI { + metrics = fx + } + if !ff.NoCommitmentAPI { + reserved = fx + } + + c, err := krds.NewCollector(fx, metrics, reserved, cfg) + if err != nil { + return nil, nil, fmt.Errorf("%s: %w", path, err) + } + snap, err := c.Collect(ctx) + if err != nil { + return nil, nil, fmt.Errorf("%s: %w", path, err) + } + + var warnings []string + if got := c.Window(); got != cfg.Window { + warnings = append(warnings, fmt.Sprintf( + "%s: observation window clamped to %s (1-minute CloudWatch datapoints live %s)", + path, got.String(), krds.RetentionAtOneMinute)) + } + for _, w := range snap.Warnings { + warnings = append(warnings, path+": "+w) + } + return snap, warnings, nil +} + +// loadRDSRates resolves the rate card. +// +// Layering, not replacement: DefaultRates().Merge(loaded) lets an operator +// supply the SQL Server and Oracle rows this package ships none of, without +// restating every open-source row. Every loaded row is stamped +// `operator-supplied` by pkg/rds and is therefore claimable; every shipped row +// is `unverified` and can size a fact and never a saving. That asymmetry is +// the intended, loud failure mode — until somebody who can see their own +// invoice supplies a file, no RDS dollar is claimable. +func loadRDSRates(path string) (krds.RateCard, error) { + base := krds.DefaultRates() + if path == "" { + return base, nil + } + over, err := krds.LoadRatesFile(path) + if err != nil { + return krds.RateCard{}, fmt.Errorf("--rds-rates %s: %w", path, err) + } + return base.Merge(over), nil +} diff --git a/cmd/kilter/rds_test.go b/cmd/kilter/rds_test.go new file mode 100644 index 0000000..b91d30a --- /dev/null +++ b/cmd/kilter/rds_test.go @@ -0,0 +1,553 @@ +package main + +import ( + "encoding/json" + "math" + "os" + "path/filepath" + "strings" + "testing" + "time" + + "github.com/agenticode/kilter/pkg/domain" + kcommit "github.com/agenticode/kilter/pkg/pricing/commit" + krds "github.com/agenticode/kilter/pkg/rds" +) + +// These tests drive pkg/rds's REAL collector and REAL domain through the REAL +// CLI entry point over a recorded account. Nothing here links an AWS SDK, +// opens a socket, reads a credential or reads a clock. + +// rdsArgs is the input the RDS tests share. +func rdsArgs(t *testing.T, sub string, extra ...string) []string { + t.Helper() + args := []string{sub, + "--now", fixtureNow.Format(time.RFC3339), + "--scope", fixtureScope, + "--region", fixtureRegion, + "--domain", "rds", + "--rds-fixture", readFixture(t, rdsFixtureFileName), + } + return append(args, extra...) +} + +// writeRDSFixture writes a modified account to a temp file and returns the path. +func writeRDSFixture(t *testing.T, f rdsFixtureFile) string { + t.Helper() + raw, err := json.Marshal(f) + if err != nil { + t.Fatal(err) + } + path := filepath.Join(t.TempDir(), "account.json") + if err := os.WriteFile(path, raw, 0o644); err != nil { + t.Fatal(err) + } + return path +} + +// TestRDSIsReachableFromTheBinary is the reason this unit exists. pkg/rds +// shipped 4,980 lines of production Go that the binary could not call. +func TestRDSIsReachableFromTheBinary(t *testing.T) { + var env reportEnvelope + runJSON(t, &env, rdsArgs(t, "report")...) + + dr, ok := env.Report.For(domain.RDS) + if !ok { + t.Fatal("the rds domain does not appear in the aggregate report") + } + if !dr.Health.Ready { + t.Errorf("rds is not ready after a collection: %s", dr.Health.Reason) + } + // Seven instances went in; every one of them came out with a reason. + if dr.Health.Targets != 7 { + t.Errorf("rds tracks %d targets, want 7", dr.Health.Targets) + } + if dr.Refused == 0 { + t.Fatal("rds produced no refusals; the refusals ARE this domain's output") + } + + // The four codes that carry this domain's whole argument must all be + // reachable from the binary, not merely present in the package. + want := map[string]bool{ + krds.ReasonInstanceClassIsAFailover: false, + krds.ReasonStorageCannotShrink: false, + krds.ReasonFreeableMemoryIsPageCache: false, + krds.ReasonAuroraNotSupported: false, + } + for _, ref := range env.Report.Refusals { + if ref.Target.Domain != domain.RDS { + continue + } + if _, ok := want[ref.Code]; ok { + want[ref.Code] = true + } + if ref.Reason == "" { + t.Errorf("%s refused %s without saying why", ref.Code, ref.Target.ID) + } + } + for code, seen := range want { + if !seen { + t.Errorf("refusal code %q never reached the CLI", code) + } + } +} + +// TestRDSProposesNothingAndClaimsNothing. +// +// pkg/rds/FINDINGS.md §2: Report.Totals.Proposals is 0 in every report this +// unit can produce, and that is the deliverable rather than an omission. The +// wiring must not manufacture one — not through the generic seam, not through +// the ledger, not through the aggregate roll-up. +func TestRDSProposesNothingAndClaimsNothing(t *testing.T) { + var env reportEnvelope + runJSON(t, &env, rdsArgs(t, "report")...) + for _, rec := range env.Report.Recommendations { + if rec.Target.Domain == domain.RDS { + t.Errorf("rds emitted a recommendation for %s; this domain proposes nothing", rec.Target.ID) + } + } + dr, _ := env.Report.For(domain.RDS) + if dr.ClaimableMonthlyUSD != 0 || dr.GrossMonthlyUSD != 0 { + t.Errorf("rds claims $%v of $%v gross; every shipped rate is unverified and cannot become a saving", + dr.ClaimableMonthlyUSD, dr.GrossMonthlyUSD) + } +} + +// TestFreeableMemoryVerdictDiffersByEngineThroughTheWiring is trap 9, asserted +// at the CLI rather than in the package that implements it. +// +// db-pg-primary and db-mysql-multiaz both have a large, complete, flat +// FreeableMemory series. PostgreSQL's is page cache that MemAvailable counts +// as available, so it is never converted into a headroom number at all; +// MySQL's is anonymous buffer-pool memory, so it IS readable and the downsize +// is refused for a different reason. A wiring that flattened the domain +// through domain.Sample would lose exactly this distinction. +func TestFreeableMemoryVerdictDiffersByEngineThroughTheWiring(t *testing.T) { + var env reportEnvelope + runJSON(t, &env, rdsArgs(t, "report")...) + + codes := map[string]map[string]bool{} + for _, ref := range env.Report.Refusals { + if ref.Target.Domain != domain.RDS { + continue + } + id := ref.Target.ID + if codes[id] == nil { + codes[id] = map[string]bool{} + } + codes[id][ref.Code] = true + } + pg, my := findRDSTarget(t, codes, "db-pg-primary"), findRDSTarget(t, codes, "db-mysql-multiaz") + + if !codes[pg][krds.ReasonFreeableMemoryIsPageCache] { + t.Errorf("PostgreSQL did not refuse with %q: %v", krds.ReasonFreeableMemoryIsPageCache, keysOf(codes[pg])) + } + if codes[pg][krds.ReasonBufferPoolScalesWithClass] { + t.Error("PostgreSQL was given MySQL's buffer-pool verdict") + } + if !codes[my][krds.ReasonBufferPoolScalesWithClass] { + t.Errorf("MySQL did not refuse with %q: %v", krds.ReasonBufferPoolScalesWithClass, keysOf(codes[my])) + } + if codes[my][krds.ReasonFreeableMemoryIsPageCache] { + t.Error("MySQL was given PostgreSQL's page-cache verdict") + } +} + +// TestMultiAZBillsTwiceThroughTheWiring: trap 10 survives the CLI. The +// Multi-AZ instance's usage line costs exactly twice the Single-AZ rate for +// the same class, and the storage line does not move. +func TestMultiAZBillsTwiceThroughTheWiring(t *testing.T) { + lines := rdsBaselineLines(t) + multi, ok := lines["arn:aws:rds:us-east-1:000000000000:db:db-mysql-multiaz/instance"] + if !ok { + t.Fatalf("the Multi-AZ instance produced no usage line: %v", keysOf(lines)) + } + single, ok := lines["arn:aws:rds:us-east-1:000000000000:db:db-pg-replica/instance"] + if !ok { + t.Fatalf("the Single-AZ instance produced no usage line: %v", keysOf(lines)) + } + if multi.InstanceType != single.InstanceType { + t.Fatalf("fixture drift: %s vs %s", multi.InstanceType, single.InstanceType) + } + // The multiplier is an exact small integer, so this holds to the last bit. + if multi.ODRate != 2*single.ODRate { + t.Errorf("Multi-AZ rate $%v, want exactly 2 × the Single-AZ $%v", multi.ODRate, single.ODRate) + } + if multi.Deployment != kcommit.RDSMultiAZInstance { + t.Errorf("deployment = %q, want %q", multi.Deployment, kcommit.RDSMultiAZInstance) + } +} + +// TestRDSUsageLinesEnterTheAccountWideBaseline. +// +// pkg/rds/FINDINGS.md §6.5 says RDS lines must be spliced into whatever the +// other domains contribute, because Compute Savings Plans absorb account-wide +// and a per-domain view over- or under-states absorption. This asserts the +// splice actually happened through the real runtime, and that only priced +// instances contributed. +func TestRDSUsageLinesEnterTheAccountWideBaseline(t *testing.T) { + lines := rdsBaselineLines(t) + if len(lines) == 0 { + t.Fatal("no RDS usage line reached the account-wide baseline") + } + for id, l := range lines { + if l.Kind != kcommit.KindRDS { + t.Errorf("%s: kind %q, want %q", id, l.Kind, kcommit.KindRDS) + } + if l.InstanceType == "" { + // commit.Usage.Validate requires a class on every KindRDS line, + // which is what structurally prevents a storage line from ever + // becoming a covered line. + t.Errorf("%s: no DB instance class", id) + } + if l.ODRate <= 0 { + t.Errorf("%s: rate $%v — an unpriced instance must not enter the baseline, "+ + "because a zero-rate line makes a reservation look like it is absorbing "+ + "usage that costs nothing", id, l.ODRate) + } + } + // Aurora, the cluster member and the mode=off instance are excluded before + // pricing, so they cannot appear. + for _, banned := range []string{"db-aurora", "db-mysql-cluster", "db-legacy", "db-mssql"} { + for id := range lines { + if strings.Contains(id, banned) { + t.Errorf("%s reached the baseline; it was never priced", banned) + } + } + } +} + +// TestNoCloudWatchPermissionIsACompleteReportNotAFailure. +// +// §6.2: a caller holding rds:Describe* and not cloudwatch:GetMetricData still +// gets a complete inventory, and every instance in it honestly refuses with +// no-metric-evidence. The wiring must deliver that report rather than an +// error, and it must NOT deliver an idle verdict manufactured out of silence. +func TestNoCloudWatchPermissionIsACompleteReportNotAFailure(t *testing.T) { + f := buildRDSFixture() + f.NoMetricsAPI = true + path := writeRDSFixture(t, f) + + var env reportEnvelope + runJSON(t, &env, "report", "--now", fixtureNow.Format(time.RFC3339), + "--scope", fixtureScope, "--region", fixtureRegion, + "--domain", "rds", "--rds-fixture", path, "--json") + + dr, ok := env.Report.For(domain.RDS) + if !ok || dr.Health.Targets != 7 { + t.Fatalf("the inventory did not survive the missing metrics permission: %+v", dr) + } + var noEvidence int + for _, ref := range env.Report.Refusals { + if ref.Code == krds.ReasonNoMetricEvidence { + noEvidence++ + } + } + if noEvidence == 0 { + t.Error("no instance refused with no-metric-evidence; silence was read as data") + } + // The idle advisory is the one that must never fire on silence. + out := run(t, "report", "--now", fixtureNow.Format(time.RFC3339), + "--scope", fixtureScope, "--region", fixtureRegion, + "--domain", "rds", "--rds-fixture", path, "--rds-detail") + if strings.Contains(out, krds.AdvisoryIdleInstance) || strings.Contains(out, krds.AdvisoryIdleReadReplica) { + t.Errorf("an idle verdict was manufactured from an unanswered CloudWatch:\n%s", out) + } +} + +// TestTheWindowIsClampedAndTheClampIsSaidOutLoud. +// +// 1-minute CloudWatch datapoints live 15 days. A 30-day request does not fail; +// it returns 15 days of data inside a 30-day window, and silence read across +// the other 15 is how "this database had no connections for a month" gets +// manufactured. The collector clamps, and the CLI says so rather than +// rendering the window the operator asked for. +func TestTheWindowIsClampedAndTheClampIsSaidOutLoud(t *testing.T) { + var env reportEnvelope + runJSON(t, &env, rdsArgs(t, "report", "--rds-window", "720h")...) + var found bool + for _, w := range env.Warnings { + if strings.Contains(w, "clamped") { + found = true + } + } + if !found { + t.Errorf("a 30-day window was accepted silently: %v", env.Warnings) + } + out := run(t, rdsArgs(t, "report", "--rds-window", "720h", "--rds-detail")...) + if strings.Contains(out, "720h0m0s window") { + t.Errorf("the report renders the REQUESTED window rather than the observed one:\n%s", out) + } + if !strings.Contains(out, "360h0m0s window") { + t.Errorf("the report does not render the clamped 15-day window:\n%s", out) + } +} + +// TestRDSPlanIsRefusedByTheCore. +// +// Not by pkg/rds being polite. Registry.PlanSteps checks Health BEFORE the +// domain is consulted, and no actuator exists for the kind, so there are two +// independent walls in front of an RDS step and the domain's own +// unconditional refusal is the third. +func TestRDSPlanIsRefusedByTheCore(t *testing.T) { + var env struct { + Plans []domain.Plan `json:"plans"` + } + runJSON(t, &env, rdsArgs(t, "plan")...) + if len(env.Plans) != 1 { + t.Fatalf("got %d plans, want 1", len(env.Plans)) + } + p := env.Plans[0] + if p.Kind != domain.RDS { + t.Fatalf("kind = %q", p.Kind) + } + if p.Actuatable { + t.Error("rds claims an actuator; there is no mutating RDS API anywhere in the tree") + } + if p.RefusalCode != domain.RefuseReportOnly { + t.Errorf("refusal = %q (%s), want %q", p.RefusalCode, p.Refusal, domain.RefuseReportOnly) + } + if len(p.Steps) != 0 { + t.Errorf("rds produced %d steps", len(p.Steps)) + } +} + +// TestRDSOutputIsShuffleInvariantAndByteIdentical. +// +// Two properties in one, because they fail differently. Repeating in ONE +// process is the real determinism test — Go randomizes map iteration on every +// range — and shuffling the recorded account is the money test: pkg/ecs +// shipped a bug this quarter where float sums varied with addend order, so a +// total that is not sorted before it is summed is not a function of its +// inputs. +func TestRDSOutputIsShuffleInvariantAndByteIdentical(t *testing.T) { + base := run(t, rdsArgs(t, "report", "--rds-detail")...) + for i := 0; i < 8; i++ { + if got := run(t, rdsArgs(t, "report", "--rds-detail")...); got != base { + t.Fatalf("run %d differs from run 0 in the same process", i) + } + } + + // Now permute everything the collector could plausibly walk in a different + // order: the inventory pages, the cluster list and the reservations. + for i, perm := range [][]int{ + {6, 5, 4, 3, 2, 1, 0}, + {3, 0, 6, 1, 5, 2, 4}, + {1, 2, 0, 4, 3, 6, 5}, + } { + f := buildRDSFixture() + shuffled := make([]krds.DBInstanceRecord, len(f.Instances)) + for j, at := range perm { + shuffled[j] = f.Instances[at] + } + f.Instances = shuffled + f.Clusters = []krds.DBClusterRecord{f.Clusters[1], f.Clusters[0]} + path := writeRDSFixture(t, f) + + got := run(t, "report", "--now", fixtureNow.Format(time.RFC3339), + "--scope", fixtureScope, "--region", fixtureRegion, + "--domain", "rds", "--rds-fixture", path, "--rds-detail") + if got != base { + t.Errorf("permutation %d changed the report; a total that depends on "+ + "input order is not a function of its inputs", i) + } + } +} + +// TestRDSSnapshotAlsoArrivesThroughTheGenericSeam. +// +// The fixture path is one of two ways in. A collector running elsewhere ships +// a domain.Snapshot with the native snapshot in Payload, and --snapshot routes +// it by its "domain" field like every other domain's. Both must produce the +// same findings, because Payload is the lossless half of the projection. +func TestRDSSnapshotAlsoArrivesThroughTheGenericSeam(t *testing.T) { + // Collect once, exactly as `kilter domains --rds-fixture` does. + snap, _, err := collectRDS(t.Context(), readFixture(t, rdsFixtureFileName), + fixtureScope, fixtureRegion, fixtureNow, 14*24*time.Hour) + if err != nil { + t.Fatal(err) + } + generic := snap.Generic() + raw, err := json.Marshal(generic) + if err != nil { + t.Fatal(err) + } + path := filepath.Join(t.TempDir(), "rds-snapshot.json") + if err := os.WriteFile(path, raw, 0o644); err != nil { + t.Fatal(err) + } + + viaSnapshot := run(t, "report", "--now", fixtureNow.Format(time.RFC3339), + "--scope", fixtureScope, "--region", fixtureRegion, + "--domain", "rds", "--snapshot", path, "--rds-detail") + viaFixture := run(t, rdsArgs(t, "report", "--rds-detail")...) + if viaSnapshot != viaFixture { + t.Errorf("the generic seam lost evidence the collector path kept:\n--- snapshot ---\n%s\n--- fixture ---\n%s", + viaSnapshot, viaFixture) + } +} + +// ---------------------------------------------------------------- helpers + +// rdsBaselineLines runs the real runtime and returns the RDS half of the +// account-wide commitment baseline, keyed by line ID. +func rdsBaselineLines(t *testing.T) map[string]kcommit.UsageLine { + t.Helper() + df := &domainFlags{ + now: fixtureNow.Format(time.RFC3339), + scope: fixtureScope, + region: fixtureRegion, + rdsWindow: 14 * 24 * time.Hour, + } + df.kinds = repeatedFlag{"rds"} + df.rdsFixtures = repeatedFlag{readFixture(t, rdsFixtureFileName)} + rt, err := buildRuntime(df) + if err != nil { + t.Fatal(err) + } + out := map[string]kcommit.UsageLine{} + for _, l := range rt.Ledger.Baseline() { + if l.Kind == kcommit.KindRDS { + out[l.ID] = l + } + } + return out +} + +func findRDSTarget(t *testing.T, codes map[string]map[string]bool, want string) string { + t.Helper() + for id := range codes { + if strings.Contains(id, want) { + return id + } + } + t.Fatalf("no refusal for %q; saw %v", want, keysOf(codes)) + return "" +} + +func keysOf[V any](m map[string]V) []string { + out := make([]string, 0, len(m)) + for k := range m { + out = append(out, k) + } + return out +} + +// ---------------------------------------------------------- adversarial + +// hostileLiner is a domain that contributes poisoned usage lines to the +// account-wide commitment baseline. +// +// It is the same attack j1-wire closed one level down. A domain's OUTPUT +// reaches a shared structure the core owns, and until this unit nothing +// checked it: the baseline is what every OTHER domain's net savings are +// computed against, so a domain that inflates it makes some other domain's +// stranded commitment look absorbed and its saving look real. +type hostileLiner struct { + domain.Domain + lines []kcommit.UsageLine +} + +func (h *hostileLiner) UsageLines(time.Time, domain.Netter) []kcommit.UsageLine { + out := make([]kcommit.UsageLine, len(h.lines)) + copy(out, h.lines) + return out +} + +// honestPart is the minimum a registry needs to hold a hostile liner. +type honestPart struct{ kind domain.Kind } + +func (h *honestPart) Kind() domain.Kind { return h.kind } +func (h *honestPart) Learn(*domain.Snapshot) error { return nil } +func (h *honestPart) Recommend(time.Time, domain.Netter) []domain.Recommendation { + return nil +} +func (h *honestPart) PlanSteps([]domain.Recommendation, domain.Guard) ([]domain.Step, error) { + return nil, domain.ErrReportOnly +} +func (h *honestPart) Health(time.Time) domain.Health { + return domain.Health{Kind: h.kind, Ready: true, ReportOnly: true} +} +func (h *honestPart) Checkpoint() ([]byte, error) { return nil, nil } +func (h *honestPart) Restore([]byte) error { return nil } + +// TestPoisonedUsageLinesNeverReachTheAccountWideBaseline. +// +// Every rejection direction is conservative: a dropped line means less usage +// to absorb a commitment, which means MORE apparent stranding and a SMALLER +// claimed saving. Under-claiming is the only safe way to be wrong about a +// number somebody puts in a business case. +func TestPoisonedUsageLinesNeverReachTheAccountWideBaseline(t *testing.T) { + good := kcommit.UsageLine{ + ID: "arn:honest/instance", Kind: kcommit.KindRDS, Region: fixtureRegion, + InstanceType: "db.r6i.large", Engine: "postgresql", + Deployment: kcommit.RDSSingleAZ, Unit: "Instance-Hours", Quantity: 1, ODRate: 0.24, + } + poisoned := []kcommit.UsageLine{ + good, + // Anonymous: never matches in the splice, so it is always appended. + func() kcommit.UsageLine { l := good; l.ID = ""; return l }(), + // Free money: prices real usage at nothing. + func() kcommit.UsageLine { l := good; l.ID = "arn:zero/instance"; l.ODRate = 0; return l }(), + func() kcommit.UsageLine { + l := good + l.ID = "arn:nan/instance" + l.ODRate = math.NaN() + return l + }(), + func() kcommit.UsageLine { l := good; l.ID = "arn:noqty/instance"; l.Quantity = 0; return l }(), + // A second line under an ID already contributed: the splice REPLACES, + // so this silently rewrites the honest one. + func() kcommit.UsageLine { l := good; l.ODRate = 999; return l }(), + } + + reg := domain.NewRegistry() + if err := reg.Register(&hostileLiner{Domain: &honestPart{kind: domain.RDS}, lines: poisoned}); err != nil { + t.Fatal(err) + } + ledger, warnings := buildLedger(reg, nil, fixtureNow) + + base := ledger.Baseline() + if len(base) != 1 { + t.Fatalf("got %d baseline lines, want only the honest one: %+v", len(base), base) + } + if base[0].ODRate != 0.24 { + t.Errorf("the honest line was rewritten to $%v", base[0].ODRate) + } + if len(warnings) != len(poisoned)-1 { + t.Errorf("got %d warnings for %d poisoned lines: %v", len(warnings), len(poisoned)-1, warnings) + } + // A drop is never silent. + for _, w := range warnings { + if !strings.Contains(w, "dropped a usage line") { + t.Errorf("warning does not say what happened: %q", w) + } + } +} + +// TestTheShippedDomainsContributeCleanBaselineLines is the control: the gate +// above must be a no-op for every domain this binary actually wires, or it is +// silently shrinking a real baseline. +func TestTheShippedDomainsContributeCleanBaselineLines(t *testing.T) { + df := &domainFlags{ + now: fixtureNow.Format(time.RFC3339), scope: fixtureScope, region: fixtureRegion, + rdsWindow: 14 * 24 * time.Hour, + } + df.snapshots = repeatedFlag{ + readFixture(t, "ec2-instances.json"), readFixture(t, "ec2-volumes.json"), + readFixture(t, "ecs-services.json"), readFixture(t, "lambda-functions.json"), + } + df.rdsFixtures = repeatedFlag{readFixture(t, rdsFixtureFileName)} + rt, err := buildRuntime(df) + if err != nil { + t.Fatal(err) + } + for _, w := range rt.Warnings { + if strings.Contains(w, "dropped a usage line") { + t.Errorf("a shipped domain's baseline line was dropped: %s", w) + } + } + if len(rt.Ledger.Baseline()) == 0 { + t.Fatal("the account-wide baseline is empty; the control proves nothing") + } +} diff --git a/cmd/kilter/rdsfixture_test.go b/cmd/kilter/rdsfixture_test.go new file mode 100644 index 0000000..eb5beb8 --- /dev/null +++ b/cmd/kilter/rdsfixture_test.go @@ -0,0 +1,225 @@ +package main + +import ( + "encoding/json" + "os" + "testing" + "time" + + krds "github.com/agenticode/kilter/pkg/rds" +) + +// The recorded RDS account, and why each row is in it. +// +// Same discipline as testdata/ec2-instances.json: generated, committed, and +// re-checked by TestWriteRDSFixture, so a regeneration that changes the bytes +// means pkg/rds's collector changed and a reviewer should look. Regenerate: +// +// go test ./cmd/kilter -run TestWriteRDSFixture -update-fixtures +// +// Every instance is the INTERESTING case rather than the easy one, chosen so +// the wiring exercises a different refusal path per row: +// +// db-pg-primary PostgreSQL, 500 GiB gp2 with autoscaling on and 80 % +// of it never used — trap 9 (FreeableMemory is page +// cache, so the series is not converted to headroom at +// all) and trap 8 (the floor has a dollar and no API). +// db-mysql-multiaz MySQL Multi-AZ — trap 10: the instance line is +// doubled and the storage line is not. Its FreeableMemory +// IS readable (anonymous buffer pool) and the downsize is +// STILL refused, for a different reason. Same metric, +// different engine, different verdict: that is the trap +// nothing else in the tree catches. +// db-pg-replica a read replica with zero connections across the whole +// window — the one replica finding that is safe to state. +// db-mssql SQL Server EE license-included — refused BY NAME with +// engine-not-priced rather than quoted an open-source +// rate for a licensed engine. +// db-aurora Aurora, refused by name (trap 16). +// db-mysql-cluster a Multi-AZ DB CLUSTER member on MySQL, refused under +// its OWN name and not Aurora's — §5.3, and the reason +// DescribeDBClusters is read at all. +// db-legacy tagged kilter.dev/mode=off — the guardrail, which +// reaches the report only because ListTagsForResource is +// wired. +// +// One Reserved DB Instance covers db.r6i.xlarge on PostgreSQL, so the +// commitment seam has something to answer with. + +const rdsFixtureFileName = "rds-account.json" + +func rdsARN(id string) string { + return "arn:aws:rds:" + fixtureRegion + ":000000000000:db:" + id +} + +// rdsSeries records one metric at a 6-hour cadence across the window. The +// cadence is deliberately coarse: every evidence gate in pkg/rds is about +// window SPAN and delivery completeness, not sample count, so a dense series +// would add a megabyte of JSON and prove nothing extra. +func rdsSeries(value float64) []krds.Point { + const step = 6 * time.Hour + const days = 14 + n := int((days * 24 * time.Hour) / step) + start := fixtureNow.Add(-days * 24 * time.Hour).Add(step) + return krds.SyntheticMetric(start, step, n, value) +} + +func buildRDSFixture() rdsFixtureFile { + const gib = float64(1 << 30) + inst := []krds.DBInstanceRecord{ + { + DBInstanceIdentifier: "db-pg-primary", DBInstanceArn: rdsARN("db-pg-primary"), + DBInstanceClass: "db.r6i.xlarge", DBInstanceStatus: krds.StatusAvailable, + Engine: "postgres", EngineVersion: "16.4", LicenseModel: krds.LicenseGPL, + AvailabilityZone: fixtureRegion + "a", + // 500 allocated, autoscaling to 1000: the floor can move on its + // own and leaves no CloudTrail event. + AllocatedStorage: 500, MaxAllocatedStorage: 1000, + StorageType: krds.StorageGP2, + ReadReplicaDBInstanceIdentifiers: []string{"db-pg-replica"}, + InstanceCreateTime: fixtureNow.Add(-400 * 24 * time.Hour), + }, + { + DBInstanceIdentifier: "db-mysql-multiaz", DBInstanceArn: rdsARN("db-mysql-multiaz"), + DBInstanceClass: "db.r6i.large", DBInstanceStatus: krds.StatusAvailable, + Engine: "mysql", EngineVersion: "8.0.39", LicenseModel: krds.LicenseGPL, + MultiAZ: true, AvailabilityZone: fixtureRegion + "b", + AllocatedStorage: 200, StorageType: krds.StorageGP3, Iops: 3000, + InstanceCreateTime: fixtureNow.Add(-300 * 24 * time.Hour), + }, + { + DBInstanceIdentifier: "db-pg-replica", DBInstanceArn: rdsARN("db-pg-replica"), + DBInstanceClass: "db.r6i.large", DBInstanceStatus: krds.StatusAvailable, + Engine: "postgres", EngineVersion: "16.4", LicenseModel: krds.LicenseGPL, + AvailabilityZone: fixtureRegion + "c", + ReadReplicaSourceDBInstanceIdentifier: "db-pg-primary", + AllocatedStorage: 500, StorageType: krds.StorageGP2, + InstanceCreateTime: fixtureNow.Add(-200 * 24 * time.Hour), + }, + { + DBInstanceIdentifier: "db-mssql", DBInstanceArn: rdsARN("db-mssql"), + DBInstanceClass: "db.r6i.xlarge", DBInstanceStatus: krds.StatusAvailable, + Engine: "sqlserver-ee", EngineVersion: "15.00", LicenseModel: krds.LicenseIncluded, + AvailabilityZone: fixtureRegion + "a", + AllocatedStorage: 300, StorageType: krds.StorageGP2, + InstanceCreateTime: fixtureNow.Add(-500 * 24 * time.Hour), + }, + { + DBInstanceIdentifier: "db-aurora", DBInstanceArn: rdsARN("db-aurora"), + DBInstanceClass: "db.r6i.large", DBInstanceStatus: krds.StatusAvailable, + Engine: "aurora-postgresql", EngineVersion: "15.4", LicenseModel: krds.LicenseGPL, + DBClusterIdentifier: "aurora-prod", AvailabilityZone: fixtureRegion + "a", + InstanceCreateTime: fixtureNow.Add(-120 * 24 * time.Hour), + }, + { + DBInstanceIdentifier: "db-mysql-cluster", DBInstanceArn: rdsARN("db-mysql-cluster"), + DBInstanceClass: "db.r6i.large", DBInstanceStatus: krds.StatusAvailable, + Engine: "mysql", EngineVersion: "8.0.39", LicenseModel: krds.LicenseGPL, + DBClusterIdentifier: "mazdb-prod", AvailabilityZone: fixtureRegion + "b", + AllocatedStorage: 100, StorageType: krds.StorageGP3, + InstanceCreateTime: fixtureNow.Add(-60 * 24 * time.Hour), + }, + { + DBInstanceIdentifier: "db-legacy", DBInstanceArn: rdsARN("db-legacy"), + DBInstanceClass: "db.t3.medium", DBInstanceStatus: krds.StatusAvailable, + Engine: "postgres", EngineVersion: "13.16", LicenseModel: krds.LicenseGPL, + AvailabilityZone: fixtureRegion + "c", + AllocatedStorage: 50, StorageType: krds.StorageGP2, + InstanceCreateTime: fixtureNow.Add(-900 * 24 * time.Hour), + }, + } + + clusters := []krds.DBClusterRecord{ + { + DBClusterIdentifier: "aurora-prod", Engine: "aurora-postgresql", + EngineMode: "provisioned", DBClusterMembers: []string{"db-aurora"}, + ServerlessV2MinCapacity: 0.5, ServerlessV2MaxCapacity: 16, + }, + { + // A PostgreSQL/MySQL Multi-AZ DB cluster. Calling it "Aurora" + // would be a false statement in a report whose whole value is + // that its statements are true. + DBClusterIdentifier: "mazdb-prod", Engine: "mysql", + EngineMode: "provisioned", DBClusterMembers: []string{"db-mysql-cluster"}, + }, + } + + tags := map[string]map[string]string{ + rdsARN("db-pg-primary"): {"env": "prod", "team": "payments"}, + rdsARN("db-legacy"): {krds.TagKilterMode: "off"}, + } + + metrics := map[string][]krds.Point{ + // A busy primary whose FreeableMemory looks like 9 GiB of headroom and + // is nothing of the kind. + "db-pg-primary/" + krds.MetricCPUUtilization: rdsSeries(28), + "db-pg-primary/" + krds.MetricFreeableMemory: rdsSeries(9 * gib), + "db-pg-primary/" + krds.MetricFreeStorageSpace: rdsSeries(400 * gib), + "db-pg-primary/" + krds.MetricDatabaseConns: rdsSeries(42), + + // The same-shaped memory series on MySQL, where it IS readable. + "db-mysql-multiaz/" + krds.MetricCPUUtilization: rdsSeries(31), + "db-mysql-multiaz/" + krds.MetricFreeableMemory: rdsSeries(6 * gib), + "db-mysql-multiaz/" + krds.MetricFreeStorageSpace: rdsSeries(150 * gib), + "db-mysql-multiaz/" + krds.MetricDatabaseConns: rdsSeries(12), + + // Zero connections and near-zero CPU across the whole window. + "db-pg-replica/" + krds.MetricCPUUtilization: rdsSeries(1), + "db-pg-replica/" + krds.MetricFreeableMemory: rdsSeries(11 * gib), + "db-pg-replica/" + krds.MetricFreeStorageSpace: rdsSeries(420 * gib), + "db-pg-replica/" + krds.MetricDatabaseConns: rdsSeries(0), + + "db-mssql/" + krds.MetricCPUUtilization: rdsSeries(15), + "db-mssql/" + krds.MetricDatabaseConns: rdsSeries(8), + } + + return rdsFixtureFile{ + Instances: inst, + Clusters: clusters, + Tags: tags, + Metrics: metrics, + Reservations: []krds.ReservedDBInstanceRecord{{ + ReservedDBInstanceId: "ri-rds-1", DBInstanceClass: "db.r6i.xlarge", + DBInstanceCount: 1, ProductDescription: "postgresql", + OfferingType: "All Upfront", State: "active", + FixedPrice: 2800, UsagePrice: 0, + Duration: int64((365 * 24 * time.Hour).Seconds()), + StartTime: fixtureNow.Add(-180 * 24 * time.Hour), + }}, + // Two pages, so the collector's pagination is actually exercised + // rather than merely present. + PageSize: 3, + } +} + +// TestWriteRDSFixture mirrors TestWriteDomainFixtures: it regenerates the +// recorded account under -update-fixtures and otherwise asserts the committed +// bytes still match. The fixture is fed to pkg/rds's REAL collector, so a diff +// here is a change in that collector. +func TestWriteRDSFixture(t *testing.T) { + want, err := json.Marshal(buildRDSFixture()) + if err != nil { + t.Fatalf("marshal: %v", err) + } + want = append(want, '\n') + path := fixturePath(rdsFixtureFileName) + if *updateFixtures { + if err := os.MkdirAll("testdata", 0o755); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(path, want, 0o644); err != nil { + t.Fatal(err) + } + t.Logf("wrote %s (%d bytes)", path, len(want)) + return + } + got, err := os.ReadFile(path) + if err != nil { + t.Fatalf("%v (run with -update-fixtures to create it)", err) + } + if string(got) != string(want) { + t.Errorf("%s is stale (%d bytes on disk, %d generated).\n"+ + "Regenerate with -update-fixtures and review the diff — a change here "+ + "means pkg/rds's collector changed.", path, len(got), len(want)) + } +} diff --git a/cmd/kilter/testdata/rds-account.json b/cmd/kilter/testdata/rds-account.json new file mode 100644 index 0000000..43aadaa --- /dev/null +++ b/cmd/kilter/testdata/rds-account.json @@ -0,0 +1 @@ +{"instances":[{"dbInstanceIdentifier":"db-pg-primary","dbInstanceArn":"arn:aws:rds:us-east-1:000000000000:db:db-pg-primary","dbInstanceClass":"db.r6i.xlarge","dbInstanceStatus":"available","engine":"postgres","engineVersion":"16.4","licenseModel":"general-public-license","availabilityZone":"us-east-1a","readReplicaDBInstanceIdentifiers":["db-pg-replica"],"allocatedStorage":500,"maxAllocatedStorage":1000,"storageType":"gp2","instanceCreateTime":"2025-07-22T12:00:00Z"},{"dbInstanceIdentifier":"db-mysql-multiaz","dbInstanceArn":"arn:aws:rds:us-east-1:000000000000:db:db-mysql-multiaz","dbInstanceClass":"db.r6i.large","dbInstanceStatus":"available","engine":"mysql","engineVersion":"8.0.39","licenseModel":"general-public-license","multiAZ":true,"availabilityZone":"us-east-1b","allocatedStorage":200,"storageType":"gp3","iops":3000,"instanceCreateTime":"2025-10-30T12:00:00Z"},{"dbInstanceIdentifier":"db-pg-replica","dbInstanceArn":"arn:aws:rds:us-east-1:000000000000:db:db-pg-replica","dbInstanceClass":"db.r6i.large","dbInstanceStatus":"available","engine":"postgres","engineVersion":"16.4","licenseModel":"general-public-license","availabilityZone":"us-east-1c","readReplicaSourceDBInstanceIdentifier":"db-pg-primary","allocatedStorage":500,"storageType":"gp2","instanceCreateTime":"2026-02-07T12:00:00Z"},{"dbInstanceIdentifier":"db-mssql","dbInstanceArn":"arn:aws:rds:us-east-1:000000000000:db:db-mssql","dbInstanceClass":"db.r6i.xlarge","dbInstanceStatus":"available","engine":"sqlserver-ee","engineVersion":"15.00","licenseModel":"license-included","availabilityZone":"us-east-1a","allocatedStorage":300,"storageType":"gp2","instanceCreateTime":"2025-04-13T12:00:00Z"},{"dbInstanceIdentifier":"db-aurora","dbInstanceArn":"arn:aws:rds:us-east-1:000000000000:db:db-aurora","dbInstanceClass":"db.r6i.large","dbInstanceStatus":"available","engine":"aurora-postgresql","engineVersion":"15.4","licenseModel":"general-public-license","dbClusterIdentifier":"aurora-prod","availabilityZone":"us-east-1a","instanceCreateTime":"2026-04-28T12:00:00Z"},{"dbInstanceIdentifier":"db-mysql-cluster","dbInstanceArn":"arn:aws:rds:us-east-1:000000000000:db:db-mysql-cluster","dbInstanceClass":"db.r6i.large","dbInstanceStatus":"available","engine":"mysql","engineVersion":"8.0.39","licenseModel":"general-public-license","dbClusterIdentifier":"mazdb-prod","availabilityZone":"us-east-1b","allocatedStorage":100,"storageType":"gp3","instanceCreateTime":"2026-06-27T12:00:00Z"},{"dbInstanceIdentifier":"db-legacy","dbInstanceArn":"arn:aws:rds:us-east-1:000000000000:db:db-legacy","dbInstanceClass":"db.t3.medium","dbInstanceStatus":"available","engine":"postgres","engineVersion":"13.16","licenseModel":"general-public-license","availabilityZone":"us-east-1c","allocatedStorage":50,"storageType":"gp2","instanceCreateTime":"2024-03-09T12:00:00Z"}],"clusters":[{"dbClusterIdentifier":"aurora-prod","engine":"aurora-postgresql","engineMode":"provisioned","dbClusterMembers":["db-aurora"],"serverlessV2MinCapacity":0.5,"serverlessV2MaxCapacity":16},{"dbClusterIdentifier":"mazdb-prod","engine":"mysql","engineMode":"provisioned","dbClusterMembers":["db-mysql-cluster"]}],"tags":{"arn:aws:rds:us-east-1:000000000000:db:db-legacy":{"kilter.dev/mode":"off"},"arn:aws:rds:us-east-1:000000000000:db:db-pg-primary":{"env":"prod","team":"payments"}},"metrics":{"db-mssql/CPUUtilization":[{"at":"2026-08-12T18:00:00Z","value":15},{"at":"2026-08-13T00:00:00Z","value":15},{"at":"2026-08-13T06:00:00Z","value":15},{"at":"2026-08-13T12:00:00Z","value":15},{"at":"2026-08-13T18:00:00Z","value":15},{"at":"2026-08-14T00:00:00Z","value":15},{"at":"2026-08-14T06:00:00Z","value":15},{"at":"2026-08-14T12:00:00Z","value":15},{"at":"2026-08-14T18:00:00Z","value":15},{"at":"2026-08-15T00:00:00Z","value":15},{"at":"2026-08-15T06:00:00Z","value":15},{"at":"2026-08-15T12:00:00Z","value":15},{"at":"2026-08-15T18:00:00Z","value":15},{"at":"2026-08-16T00:00:00Z","value":15},{"at":"2026-08-16T06:00:00Z","value":15},{"at":"2026-08-16T12:00:00Z","value":15},{"at":"2026-08-16T18:00:00Z","value":15},{"at":"2026-08-17T00:00:00Z","value":15},{"at":"2026-08-17T06:00:00Z","value":15},{"at":"2026-08-17T12:00:00Z","value":15},{"at":"2026-08-17T18:00:00Z","value":15},{"at":"2026-08-18T00:00:00Z","value":15},{"at":"2026-08-18T06:00:00Z","value":15},{"at":"2026-08-18T12:00:00Z","value":15},{"at":"2026-08-18T18:00:00Z","value":15},{"at":"2026-08-19T00:00:00Z","value":15},{"at":"2026-08-19T06:00:00Z","value":15},{"at":"2026-08-19T12:00:00Z","value":15},{"at":"2026-08-19T18:00:00Z","value":15},{"at":"2026-08-20T00:00:00Z","value":15},{"at":"2026-08-20T06:00:00Z","value":15},{"at":"2026-08-20T12:00:00Z","value":15},{"at":"2026-08-20T18:00:00Z","value":15},{"at":"2026-08-21T00:00:00Z","value":15},{"at":"2026-08-21T06:00:00Z","value":15},{"at":"2026-08-21T12:00:00Z","value":15},{"at":"2026-08-21T18:00:00Z","value":15},{"at":"2026-08-22T00:00:00Z","value":15},{"at":"2026-08-22T06:00:00Z","value":15},{"at":"2026-08-22T12:00:00Z","value":15},{"at":"2026-08-22T18:00:00Z","value":15},{"at":"2026-08-23T00:00:00Z","value":15},{"at":"2026-08-23T06:00:00Z","value":15},{"at":"2026-08-23T12:00:00Z","value":15},{"at":"2026-08-23T18:00:00Z","value":15},{"at":"2026-08-24T00:00:00Z","value":15},{"at":"2026-08-24T06:00:00Z","value":15},{"at":"2026-08-24T12:00:00Z","value":15},{"at":"2026-08-24T18:00:00Z","value":15},{"at":"2026-08-25T00:00:00Z","value":15},{"at":"2026-08-25T06:00:00Z","value":15},{"at":"2026-08-25T12:00:00Z","value":15},{"at":"2026-08-25T18:00:00Z","value":15},{"at":"2026-08-26T00:00:00Z","value":15},{"at":"2026-08-26T06:00:00Z","value":15},{"at":"2026-08-26T12:00:00Z","value":15}],"db-mssql/DatabaseConnections":[{"at":"2026-08-12T18:00:00Z","value":8},{"at":"2026-08-13T00:00:00Z","value":8},{"at":"2026-08-13T06:00:00Z","value":8},{"at":"2026-08-13T12:00:00Z","value":8},{"at":"2026-08-13T18:00:00Z","value":8},{"at":"2026-08-14T00:00:00Z","value":8},{"at":"2026-08-14T06:00:00Z","value":8},{"at":"2026-08-14T12:00:00Z","value":8},{"at":"2026-08-14T18:00:00Z","value":8},{"at":"2026-08-15T00:00:00Z","value":8},{"at":"2026-08-15T06:00:00Z","value":8},{"at":"2026-08-15T12:00:00Z","value":8},{"at":"2026-08-15T18:00:00Z","value":8},{"at":"2026-08-16T00:00:00Z","value":8},{"at":"2026-08-16T06:00:00Z","value":8},{"at":"2026-08-16T12:00:00Z","value":8},{"at":"2026-08-16T18:00:00Z","value":8},{"at":"2026-08-17T00:00:00Z","value":8},{"at":"2026-08-17T06:00:00Z","value":8},{"at":"2026-08-17T12:00:00Z","value":8},{"at":"2026-08-17T18:00:00Z","value":8},{"at":"2026-08-18T00:00:00Z","value":8},{"at":"2026-08-18T06:00:00Z","value":8},{"at":"2026-08-18T12:00:00Z","value":8},{"at":"2026-08-18T18:00:00Z","value":8},{"at":"2026-08-19T00:00:00Z","value":8},{"at":"2026-08-19T06:00:00Z","value":8},{"at":"2026-08-19T12:00:00Z","value":8},{"at":"2026-08-19T18:00:00Z","value":8},{"at":"2026-08-20T00:00:00Z","value":8},{"at":"2026-08-20T06:00:00Z","value":8},{"at":"2026-08-20T12:00:00Z","value":8},{"at":"2026-08-20T18:00:00Z","value":8},{"at":"2026-08-21T00:00:00Z","value":8},{"at":"2026-08-21T06:00:00Z","value":8},{"at":"2026-08-21T12:00:00Z","value":8},{"at":"2026-08-21T18:00:00Z","value":8},{"at":"2026-08-22T00:00:00Z","value":8},{"at":"2026-08-22T06:00:00Z","value":8},{"at":"2026-08-22T12:00:00Z","value":8},{"at":"2026-08-22T18:00:00Z","value":8},{"at":"2026-08-23T00:00:00Z","value":8},{"at":"2026-08-23T06:00:00Z","value":8},{"at":"2026-08-23T12:00:00Z","value":8},{"at":"2026-08-23T18:00:00Z","value":8},{"at":"2026-08-24T00:00:00Z","value":8},{"at":"2026-08-24T06:00:00Z","value":8},{"at":"2026-08-24T12:00:00Z","value":8},{"at":"2026-08-24T18:00:00Z","value":8},{"at":"2026-08-25T00:00:00Z","value":8},{"at":"2026-08-25T06:00:00Z","value":8},{"at":"2026-08-25T12:00:00Z","value":8},{"at":"2026-08-25T18:00:00Z","value":8},{"at":"2026-08-26T00:00:00Z","value":8},{"at":"2026-08-26T06:00:00Z","value":8},{"at":"2026-08-26T12:00:00Z","value":8}],"db-mysql-multiaz/CPUUtilization":[{"at":"2026-08-12T18:00:00Z","value":31},{"at":"2026-08-13T00:00:00Z","value":31},{"at":"2026-08-13T06:00:00Z","value":31},{"at":"2026-08-13T12:00:00Z","value":31},{"at":"2026-08-13T18:00:00Z","value":31},{"at":"2026-08-14T00:00:00Z","value":31},{"at":"2026-08-14T06:00:00Z","value":31},{"at":"2026-08-14T12:00:00Z","value":31},{"at":"2026-08-14T18:00:00Z","value":31},{"at":"2026-08-15T00:00:00Z","value":31},{"at":"2026-08-15T06:00:00Z","value":31},{"at":"2026-08-15T12:00:00Z","value":31},{"at":"2026-08-15T18:00:00Z","value":31},{"at":"2026-08-16T00:00:00Z","value":31},{"at":"2026-08-16T06:00:00Z","value":31},{"at":"2026-08-16T12:00:00Z","value":31},{"at":"2026-08-16T18:00:00Z","value":31},{"at":"2026-08-17T00:00:00Z","value":31},{"at":"2026-08-17T06:00:00Z","value":31},{"at":"2026-08-17T12:00:00Z","value":31},{"at":"2026-08-17T18:00:00Z","value":31},{"at":"2026-08-18T00:00:00Z","value":31},{"at":"2026-08-18T06:00:00Z","value":31},{"at":"2026-08-18T12:00:00Z","value":31},{"at":"2026-08-18T18:00:00Z","value":31},{"at":"2026-08-19T00:00:00Z","value":31},{"at":"2026-08-19T06:00:00Z","value":31},{"at":"2026-08-19T12:00:00Z","value":31},{"at":"2026-08-19T18:00:00Z","value":31},{"at":"2026-08-20T00:00:00Z","value":31},{"at":"2026-08-20T06:00:00Z","value":31},{"at":"2026-08-20T12:00:00Z","value":31},{"at":"2026-08-20T18:00:00Z","value":31},{"at":"2026-08-21T00:00:00Z","value":31},{"at":"2026-08-21T06:00:00Z","value":31},{"at":"2026-08-21T12:00:00Z","value":31},{"at":"2026-08-21T18:00:00Z","value":31},{"at":"2026-08-22T00:00:00Z","value":31},{"at":"2026-08-22T06:00:00Z","value":31},{"at":"2026-08-22T12:00:00Z","value":31},{"at":"2026-08-22T18:00:00Z","value":31},{"at":"2026-08-23T00:00:00Z","value":31},{"at":"2026-08-23T06:00:00Z","value":31},{"at":"2026-08-23T12:00:00Z","value":31},{"at":"2026-08-23T18:00:00Z","value":31},{"at":"2026-08-24T00:00:00Z","value":31},{"at":"2026-08-24T06:00:00Z","value":31},{"at":"2026-08-24T12:00:00Z","value":31},{"at":"2026-08-24T18:00:00Z","value":31},{"at":"2026-08-25T00:00:00Z","value":31},{"at":"2026-08-25T06:00:00Z","value":31},{"at":"2026-08-25T12:00:00Z","value":31},{"at":"2026-08-25T18:00:00Z","value":31},{"at":"2026-08-26T00:00:00Z","value":31},{"at":"2026-08-26T06:00:00Z","value":31},{"at":"2026-08-26T12:00:00Z","value":31}],"db-mysql-multiaz/DatabaseConnections":[{"at":"2026-08-12T18:00:00Z","value":12},{"at":"2026-08-13T00:00:00Z","value":12},{"at":"2026-08-13T06:00:00Z","value":12},{"at":"2026-08-13T12:00:00Z","value":12},{"at":"2026-08-13T18:00:00Z","value":12},{"at":"2026-08-14T00:00:00Z","value":12},{"at":"2026-08-14T06:00:00Z","value":12},{"at":"2026-08-14T12:00:00Z","value":12},{"at":"2026-08-14T18:00:00Z","value":12},{"at":"2026-08-15T00:00:00Z","value":12},{"at":"2026-08-15T06:00:00Z","value":12},{"at":"2026-08-15T12:00:00Z","value":12},{"at":"2026-08-15T18:00:00Z","value":12},{"at":"2026-08-16T00:00:00Z","value":12},{"at":"2026-08-16T06:00:00Z","value":12},{"at":"2026-08-16T12:00:00Z","value":12},{"at":"2026-08-16T18:00:00Z","value":12},{"at":"2026-08-17T00:00:00Z","value":12},{"at":"2026-08-17T06:00:00Z","value":12},{"at":"2026-08-17T12:00:00Z","value":12},{"at":"2026-08-17T18:00:00Z","value":12},{"at":"2026-08-18T00:00:00Z","value":12},{"at":"2026-08-18T06:00:00Z","value":12},{"at":"2026-08-18T12:00:00Z","value":12},{"at":"2026-08-18T18:00:00Z","value":12},{"at":"2026-08-19T00:00:00Z","value":12},{"at":"2026-08-19T06:00:00Z","value":12},{"at":"2026-08-19T12:00:00Z","value":12},{"at":"2026-08-19T18:00:00Z","value":12},{"at":"2026-08-20T00:00:00Z","value":12},{"at":"2026-08-20T06:00:00Z","value":12},{"at":"2026-08-20T12:00:00Z","value":12},{"at":"2026-08-20T18:00:00Z","value":12},{"at":"2026-08-21T00:00:00Z","value":12},{"at":"2026-08-21T06:00:00Z","value":12},{"at":"2026-08-21T12:00:00Z","value":12},{"at":"2026-08-21T18:00:00Z","value":12},{"at":"2026-08-22T00:00:00Z","value":12},{"at":"2026-08-22T06:00:00Z","value":12},{"at":"2026-08-22T12:00:00Z","value":12},{"at":"2026-08-22T18:00:00Z","value":12},{"at":"2026-08-23T00:00:00Z","value":12},{"at":"2026-08-23T06:00:00Z","value":12},{"at":"2026-08-23T12:00:00Z","value":12},{"at":"2026-08-23T18:00:00Z","value":12},{"at":"2026-08-24T00:00:00Z","value":12},{"at":"2026-08-24T06:00:00Z","value":12},{"at":"2026-08-24T12:00:00Z","value":12},{"at":"2026-08-24T18:00:00Z","value":12},{"at":"2026-08-25T00:00:00Z","value":12},{"at":"2026-08-25T06:00:00Z","value":12},{"at":"2026-08-25T12:00:00Z","value":12},{"at":"2026-08-25T18:00:00Z","value":12},{"at":"2026-08-26T00:00:00Z","value":12},{"at":"2026-08-26T06:00:00Z","value":12},{"at":"2026-08-26T12:00:00Z","value":12}],"db-mysql-multiaz/FreeStorageSpace":[{"at":"2026-08-12T18:00:00Z","value":161061273600},{"at":"2026-08-13T00:00:00Z","value":161061273600},{"at":"2026-08-13T06:00:00Z","value":161061273600},{"at":"2026-08-13T12:00:00Z","value":161061273600},{"at":"2026-08-13T18:00:00Z","value":161061273600},{"at":"2026-08-14T00:00:00Z","value":161061273600},{"at":"2026-08-14T06:00:00Z","value":161061273600},{"at":"2026-08-14T12:00:00Z","value":161061273600},{"at":"2026-08-14T18:00:00Z","value":161061273600},{"at":"2026-08-15T00:00:00Z","value":161061273600},{"at":"2026-08-15T06:00:00Z","value":161061273600},{"at":"2026-08-15T12:00:00Z","value":161061273600},{"at":"2026-08-15T18:00:00Z","value":161061273600},{"at":"2026-08-16T00:00:00Z","value":161061273600},{"at":"2026-08-16T06:00:00Z","value":161061273600},{"at":"2026-08-16T12:00:00Z","value":161061273600},{"at":"2026-08-16T18:00:00Z","value":161061273600},{"at":"2026-08-17T00:00:00Z","value":161061273600},{"at":"2026-08-17T06:00:00Z","value":161061273600},{"at":"2026-08-17T12:00:00Z","value":161061273600},{"at":"2026-08-17T18:00:00Z","value":161061273600},{"at":"2026-08-18T00:00:00Z","value":161061273600},{"at":"2026-08-18T06:00:00Z","value":161061273600},{"at":"2026-08-18T12:00:00Z","value":161061273600},{"at":"2026-08-18T18:00:00Z","value":161061273600},{"at":"2026-08-19T00:00:00Z","value":161061273600},{"at":"2026-08-19T06:00:00Z","value":161061273600},{"at":"2026-08-19T12:00:00Z","value":161061273600},{"at":"2026-08-19T18:00:00Z","value":161061273600},{"at":"2026-08-20T00:00:00Z","value":161061273600},{"at":"2026-08-20T06:00:00Z","value":161061273600},{"at":"2026-08-20T12:00:00Z","value":161061273600},{"at":"2026-08-20T18:00:00Z","value":161061273600},{"at":"2026-08-21T00:00:00Z","value":161061273600},{"at":"2026-08-21T06:00:00Z","value":161061273600},{"at":"2026-08-21T12:00:00Z","value":161061273600},{"at":"2026-08-21T18:00:00Z","value":161061273600},{"at":"2026-08-22T00:00:00Z","value":161061273600},{"at":"2026-08-22T06:00:00Z","value":161061273600},{"at":"2026-08-22T12:00:00Z","value":161061273600},{"at":"2026-08-22T18:00:00Z","value":161061273600},{"at":"2026-08-23T00:00:00Z","value":161061273600},{"at":"2026-08-23T06:00:00Z","value":161061273600},{"at":"2026-08-23T12:00:00Z","value":161061273600},{"at":"2026-08-23T18:00:00Z","value":161061273600},{"at":"2026-08-24T00:00:00Z","value":161061273600},{"at":"2026-08-24T06:00:00Z","value":161061273600},{"at":"2026-08-24T12:00:00Z","value":161061273600},{"at":"2026-08-24T18:00:00Z","value":161061273600},{"at":"2026-08-25T00:00:00Z","value":161061273600},{"at":"2026-08-25T06:00:00Z","value":161061273600},{"at":"2026-08-25T12:00:00Z","value":161061273600},{"at":"2026-08-25T18:00:00Z","value":161061273600},{"at":"2026-08-26T00:00:00Z","value":161061273600},{"at":"2026-08-26T06:00:00Z","value":161061273600},{"at":"2026-08-26T12:00:00Z","value":161061273600}],"db-mysql-multiaz/FreeableMemory":[{"at":"2026-08-12T18:00:00Z","value":6442450944},{"at":"2026-08-13T00:00:00Z","value":6442450944},{"at":"2026-08-13T06:00:00Z","value":6442450944},{"at":"2026-08-13T12:00:00Z","value":6442450944},{"at":"2026-08-13T18:00:00Z","value":6442450944},{"at":"2026-08-14T00:00:00Z","value":6442450944},{"at":"2026-08-14T06:00:00Z","value":6442450944},{"at":"2026-08-14T12:00:00Z","value":6442450944},{"at":"2026-08-14T18:00:00Z","value":6442450944},{"at":"2026-08-15T00:00:00Z","value":6442450944},{"at":"2026-08-15T06:00:00Z","value":6442450944},{"at":"2026-08-15T12:00:00Z","value":6442450944},{"at":"2026-08-15T18:00:00Z","value":6442450944},{"at":"2026-08-16T00:00:00Z","value":6442450944},{"at":"2026-08-16T06:00:00Z","value":6442450944},{"at":"2026-08-16T12:00:00Z","value":6442450944},{"at":"2026-08-16T18:00:00Z","value":6442450944},{"at":"2026-08-17T00:00:00Z","value":6442450944},{"at":"2026-08-17T06:00:00Z","value":6442450944},{"at":"2026-08-17T12:00:00Z","value":6442450944},{"at":"2026-08-17T18:00:00Z","value":6442450944},{"at":"2026-08-18T00:00:00Z","value":6442450944},{"at":"2026-08-18T06:00:00Z","value":6442450944},{"at":"2026-08-18T12:00:00Z","value":6442450944},{"at":"2026-08-18T18:00:00Z","value":6442450944},{"at":"2026-08-19T00:00:00Z","value":6442450944},{"at":"2026-08-19T06:00:00Z","value":6442450944},{"at":"2026-08-19T12:00:00Z","value":6442450944},{"at":"2026-08-19T18:00:00Z","value":6442450944},{"at":"2026-08-20T00:00:00Z","value":6442450944},{"at":"2026-08-20T06:00:00Z","value":6442450944},{"at":"2026-08-20T12:00:00Z","value":6442450944},{"at":"2026-08-20T18:00:00Z","value":6442450944},{"at":"2026-08-21T00:00:00Z","value":6442450944},{"at":"2026-08-21T06:00:00Z","value":6442450944},{"at":"2026-08-21T12:00:00Z","value":6442450944},{"at":"2026-08-21T18:00:00Z","value":6442450944},{"at":"2026-08-22T00:00:00Z","value":6442450944},{"at":"2026-08-22T06:00:00Z","value":6442450944},{"at":"2026-08-22T12:00:00Z","value":6442450944},{"at":"2026-08-22T18:00:00Z","value":6442450944},{"at":"2026-08-23T00:00:00Z","value":6442450944},{"at":"2026-08-23T06:00:00Z","value":6442450944},{"at":"2026-08-23T12:00:00Z","value":6442450944},{"at":"2026-08-23T18:00:00Z","value":6442450944},{"at":"2026-08-24T00:00:00Z","value":6442450944},{"at":"2026-08-24T06:00:00Z","value":6442450944},{"at":"2026-08-24T12:00:00Z","value":6442450944},{"at":"2026-08-24T18:00:00Z","value":6442450944},{"at":"2026-08-25T00:00:00Z","value":6442450944},{"at":"2026-08-25T06:00:00Z","value":6442450944},{"at":"2026-08-25T12:00:00Z","value":6442450944},{"at":"2026-08-25T18:00:00Z","value":6442450944},{"at":"2026-08-26T00:00:00Z","value":6442450944},{"at":"2026-08-26T06:00:00Z","value":6442450944},{"at":"2026-08-26T12:00:00Z","value":6442450944}],"db-pg-primary/CPUUtilization":[{"at":"2026-08-12T18:00:00Z","value":28},{"at":"2026-08-13T00:00:00Z","value":28},{"at":"2026-08-13T06:00:00Z","value":28},{"at":"2026-08-13T12:00:00Z","value":28},{"at":"2026-08-13T18:00:00Z","value":28},{"at":"2026-08-14T00:00:00Z","value":28},{"at":"2026-08-14T06:00:00Z","value":28},{"at":"2026-08-14T12:00:00Z","value":28},{"at":"2026-08-14T18:00:00Z","value":28},{"at":"2026-08-15T00:00:00Z","value":28},{"at":"2026-08-15T06:00:00Z","value":28},{"at":"2026-08-15T12:00:00Z","value":28},{"at":"2026-08-15T18:00:00Z","value":28},{"at":"2026-08-16T00:00:00Z","value":28},{"at":"2026-08-16T06:00:00Z","value":28},{"at":"2026-08-16T12:00:00Z","value":28},{"at":"2026-08-16T18:00:00Z","value":28},{"at":"2026-08-17T00:00:00Z","value":28},{"at":"2026-08-17T06:00:00Z","value":28},{"at":"2026-08-17T12:00:00Z","value":28},{"at":"2026-08-17T18:00:00Z","value":28},{"at":"2026-08-18T00:00:00Z","value":28},{"at":"2026-08-18T06:00:00Z","value":28},{"at":"2026-08-18T12:00:00Z","value":28},{"at":"2026-08-18T18:00:00Z","value":28},{"at":"2026-08-19T00:00:00Z","value":28},{"at":"2026-08-19T06:00:00Z","value":28},{"at":"2026-08-19T12:00:00Z","value":28},{"at":"2026-08-19T18:00:00Z","value":28},{"at":"2026-08-20T00:00:00Z","value":28},{"at":"2026-08-20T06:00:00Z","value":28},{"at":"2026-08-20T12:00:00Z","value":28},{"at":"2026-08-20T18:00:00Z","value":28},{"at":"2026-08-21T00:00:00Z","value":28},{"at":"2026-08-21T06:00:00Z","value":28},{"at":"2026-08-21T12:00:00Z","value":28},{"at":"2026-08-21T18:00:00Z","value":28},{"at":"2026-08-22T00:00:00Z","value":28},{"at":"2026-08-22T06:00:00Z","value":28},{"at":"2026-08-22T12:00:00Z","value":28},{"at":"2026-08-22T18:00:00Z","value":28},{"at":"2026-08-23T00:00:00Z","value":28},{"at":"2026-08-23T06:00:00Z","value":28},{"at":"2026-08-23T12:00:00Z","value":28},{"at":"2026-08-23T18:00:00Z","value":28},{"at":"2026-08-24T00:00:00Z","value":28},{"at":"2026-08-24T06:00:00Z","value":28},{"at":"2026-08-24T12:00:00Z","value":28},{"at":"2026-08-24T18:00:00Z","value":28},{"at":"2026-08-25T00:00:00Z","value":28},{"at":"2026-08-25T06:00:00Z","value":28},{"at":"2026-08-25T12:00:00Z","value":28},{"at":"2026-08-25T18:00:00Z","value":28},{"at":"2026-08-26T00:00:00Z","value":28},{"at":"2026-08-26T06:00:00Z","value":28},{"at":"2026-08-26T12:00:00Z","value":28}],"db-pg-primary/DatabaseConnections":[{"at":"2026-08-12T18:00:00Z","value":42},{"at":"2026-08-13T00:00:00Z","value":42},{"at":"2026-08-13T06:00:00Z","value":42},{"at":"2026-08-13T12:00:00Z","value":42},{"at":"2026-08-13T18:00:00Z","value":42},{"at":"2026-08-14T00:00:00Z","value":42},{"at":"2026-08-14T06:00:00Z","value":42},{"at":"2026-08-14T12:00:00Z","value":42},{"at":"2026-08-14T18:00:00Z","value":42},{"at":"2026-08-15T00:00:00Z","value":42},{"at":"2026-08-15T06:00:00Z","value":42},{"at":"2026-08-15T12:00:00Z","value":42},{"at":"2026-08-15T18:00:00Z","value":42},{"at":"2026-08-16T00:00:00Z","value":42},{"at":"2026-08-16T06:00:00Z","value":42},{"at":"2026-08-16T12:00:00Z","value":42},{"at":"2026-08-16T18:00:00Z","value":42},{"at":"2026-08-17T00:00:00Z","value":42},{"at":"2026-08-17T06:00:00Z","value":42},{"at":"2026-08-17T12:00:00Z","value":42},{"at":"2026-08-17T18:00:00Z","value":42},{"at":"2026-08-18T00:00:00Z","value":42},{"at":"2026-08-18T06:00:00Z","value":42},{"at":"2026-08-18T12:00:00Z","value":42},{"at":"2026-08-18T18:00:00Z","value":42},{"at":"2026-08-19T00:00:00Z","value":42},{"at":"2026-08-19T06:00:00Z","value":42},{"at":"2026-08-19T12:00:00Z","value":42},{"at":"2026-08-19T18:00:00Z","value":42},{"at":"2026-08-20T00:00:00Z","value":42},{"at":"2026-08-20T06:00:00Z","value":42},{"at":"2026-08-20T12:00:00Z","value":42},{"at":"2026-08-20T18:00:00Z","value":42},{"at":"2026-08-21T00:00:00Z","value":42},{"at":"2026-08-21T06:00:00Z","value":42},{"at":"2026-08-21T12:00:00Z","value":42},{"at":"2026-08-21T18:00:00Z","value":42},{"at":"2026-08-22T00:00:00Z","value":42},{"at":"2026-08-22T06:00:00Z","value":42},{"at":"2026-08-22T12:00:00Z","value":42},{"at":"2026-08-22T18:00:00Z","value":42},{"at":"2026-08-23T00:00:00Z","value":42},{"at":"2026-08-23T06:00:00Z","value":42},{"at":"2026-08-23T12:00:00Z","value":42},{"at":"2026-08-23T18:00:00Z","value":42},{"at":"2026-08-24T00:00:00Z","value":42},{"at":"2026-08-24T06:00:00Z","value":42},{"at":"2026-08-24T12:00:00Z","value":42},{"at":"2026-08-24T18:00:00Z","value":42},{"at":"2026-08-25T00:00:00Z","value":42},{"at":"2026-08-25T06:00:00Z","value":42},{"at":"2026-08-25T12:00:00Z","value":42},{"at":"2026-08-25T18:00:00Z","value":42},{"at":"2026-08-26T00:00:00Z","value":42},{"at":"2026-08-26T06:00:00Z","value":42},{"at":"2026-08-26T12:00:00Z","value":42}],"db-pg-primary/FreeStorageSpace":[{"at":"2026-08-12T18:00:00Z","value":429496729600},{"at":"2026-08-13T00:00:00Z","value":429496729600},{"at":"2026-08-13T06:00:00Z","value":429496729600},{"at":"2026-08-13T12:00:00Z","value":429496729600},{"at":"2026-08-13T18:00:00Z","value":429496729600},{"at":"2026-08-14T00:00:00Z","value":429496729600},{"at":"2026-08-14T06:00:00Z","value":429496729600},{"at":"2026-08-14T12:00:00Z","value":429496729600},{"at":"2026-08-14T18:00:00Z","value":429496729600},{"at":"2026-08-15T00:00:00Z","value":429496729600},{"at":"2026-08-15T06:00:00Z","value":429496729600},{"at":"2026-08-15T12:00:00Z","value":429496729600},{"at":"2026-08-15T18:00:00Z","value":429496729600},{"at":"2026-08-16T00:00:00Z","value":429496729600},{"at":"2026-08-16T06:00:00Z","value":429496729600},{"at":"2026-08-16T12:00:00Z","value":429496729600},{"at":"2026-08-16T18:00:00Z","value":429496729600},{"at":"2026-08-17T00:00:00Z","value":429496729600},{"at":"2026-08-17T06:00:00Z","value":429496729600},{"at":"2026-08-17T12:00:00Z","value":429496729600},{"at":"2026-08-17T18:00:00Z","value":429496729600},{"at":"2026-08-18T00:00:00Z","value":429496729600},{"at":"2026-08-18T06:00:00Z","value":429496729600},{"at":"2026-08-18T12:00:00Z","value":429496729600},{"at":"2026-08-18T18:00:00Z","value":429496729600},{"at":"2026-08-19T00:00:00Z","value":429496729600},{"at":"2026-08-19T06:00:00Z","value":429496729600},{"at":"2026-08-19T12:00:00Z","value":429496729600},{"at":"2026-08-19T18:00:00Z","value":429496729600},{"at":"2026-08-20T00:00:00Z","value":429496729600},{"at":"2026-08-20T06:00:00Z","value":429496729600},{"at":"2026-08-20T12:00:00Z","value":429496729600},{"at":"2026-08-20T18:00:00Z","value":429496729600},{"at":"2026-08-21T00:00:00Z","value":429496729600},{"at":"2026-08-21T06:00:00Z","value":429496729600},{"at":"2026-08-21T12:00:00Z","value":429496729600},{"at":"2026-08-21T18:00:00Z","value":429496729600},{"at":"2026-08-22T00:00:00Z","value":429496729600},{"at":"2026-08-22T06:00:00Z","value":429496729600},{"at":"2026-08-22T12:00:00Z","value":429496729600},{"at":"2026-08-22T18:00:00Z","value":429496729600},{"at":"2026-08-23T00:00:00Z","value":429496729600},{"at":"2026-08-23T06:00:00Z","value":429496729600},{"at":"2026-08-23T12:00:00Z","value":429496729600},{"at":"2026-08-23T18:00:00Z","value":429496729600},{"at":"2026-08-24T00:00:00Z","value":429496729600},{"at":"2026-08-24T06:00:00Z","value":429496729600},{"at":"2026-08-24T12:00:00Z","value":429496729600},{"at":"2026-08-24T18:00:00Z","value":429496729600},{"at":"2026-08-25T00:00:00Z","value":429496729600},{"at":"2026-08-25T06:00:00Z","value":429496729600},{"at":"2026-08-25T12:00:00Z","value":429496729600},{"at":"2026-08-25T18:00:00Z","value":429496729600},{"at":"2026-08-26T00:00:00Z","value":429496729600},{"at":"2026-08-26T06:00:00Z","value":429496729600},{"at":"2026-08-26T12:00:00Z","value":429496729600}],"db-pg-primary/FreeableMemory":[{"at":"2026-08-12T18:00:00Z","value":9663676416},{"at":"2026-08-13T00:00:00Z","value":9663676416},{"at":"2026-08-13T06:00:00Z","value":9663676416},{"at":"2026-08-13T12:00:00Z","value":9663676416},{"at":"2026-08-13T18:00:00Z","value":9663676416},{"at":"2026-08-14T00:00:00Z","value":9663676416},{"at":"2026-08-14T06:00:00Z","value":9663676416},{"at":"2026-08-14T12:00:00Z","value":9663676416},{"at":"2026-08-14T18:00:00Z","value":9663676416},{"at":"2026-08-15T00:00:00Z","value":9663676416},{"at":"2026-08-15T06:00:00Z","value":9663676416},{"at":"2026-08-15T12:00:00Z","value":9663676416},{"at":"2026-08-15T18:00:00Z","value":9663676416},{"at":"2026-08-16T00:00:00Z","value":9663676416},{"at":"2026-08-16T06:00:00Z","value":9663676416},{"at":"2026-08-16T12:00:00Z","value":9663676416},{"at":"2026-08-16T18:00:00Z","value":9663676416},{"at":"2026-08-17T00:00:00Z","value":9663676416},{"at":"2026-08-17T06:00:00Z","value":9663676416},{"at":"2026-08-17T12:00:00Z","value":9663676416},{"at":"2026-08-17T18:00:00Z","value":9663676416},{"at":"2026-08-18T00:00:00Z","value":9663676416},{"at":"2026-08-18T06:00:00Z","value":9663676416},{"at":"2026-08-18T12:00:00Z","value":9663676416},{"at":"2026-08-18T18:00:00Z","value":9663676416},{"at":"2026-08-19T00:00:00Z","value":9663676416},{"at":"2026-08-19T06:00:00Z","value":9663676416},{"at":"2026-08-19T12:00:00Z","value":9663676416},{"at":"2026-08-19T18:00:00Z","value":9663676416},{"at":"2026-08-20T00:00:00Z","value":9663676416},{"at":"2026-08-20T06:00:00Z","value":9663676416},{"at":"2026-08-20T12:00:00Z","value":9663676416},{"at":"2026-08-20T18:00:00Z","value":9663676416},{"at":"2026-08-21T00:00:00Z","value":9663676416},{"at":"2026-08-21T06:00:00Z","value":9663676416},{"at":"2026-08-21T12:00:00Z","value":9663676416},{"at":"2026-08-21T18:00:00Z","value":9663676416},{"at":"2026-08-22T00:00:00Z","value":9663676416},{"at":"2026-08-22T06:00:00Z","value":9663676416},{"at":"2026-08-22T12:00:00Z","value":9663676416},{"at":"2026-08-22T18:00:00Z","value":9663676416},{"at":"2026-08-23T00:00:00Z","value":9663676416},{"at":"2026-08-23T06:00:00Z","value":9663676416},{"at":"2026-08-23T12:00:00Z","value":9663676416},{"at":"2026-08-23T18:00:00Z","value":9663676416},{"at":"2026-08-24T00:00:00Z","value":9663676416},{"at":"2026-08-24T06:00:00Z","value":9663676416},{"at":"2026-08-24T12:00:00Z","value":9663676416},{"at":"2026-08-24T18:00:00Z","value":9663676416},{"at":"2026-08-25T00:00:00Z","value":9663676416},{"at":"2026-08-25T06:00:00Z","value":9663676416},{"at":"2026-08-25T12:00:00Z","value":9663676416},{"at":"2026-08-25T18:00:00Z","value":9663676416},{"at":"2026-08-26T00:00:00Z","value":9663676416},{"at":"2026-08-26T06:00:00Z","value":9663676416},{"at":"2026-08-26T12:00:00Z","value":9663676416}],"db-pg-replica/CPUUtilization":[{"at":"2026-08-12T18:00:00Z","value":1},{"at":"2026-08-13T00:00:00Z","value":1},{"at":"2026-08-13T06:00:00Z","value":1},{"at":"2026-08-13T12:00:00Z","value":1},{"at":"2026-08-13T18:00:00Z","value":1},{"at":"2026-08-14T00:00:00Z","value":1},{"at":"2026-08-14T06:00:00Z","value":1},{"at":"2026-08-14T12:00:00Z","value":1},{"at":"2026-08-14T18:00:00Z","value":1},{"at":"2026-08-15T00:00:00Z","value":1},{"at":"2026-08-15T06:00:00Z","value":1},{"at":"2026-08-15T12:00:00Z","value":1},{"at":"2026-08-15T18:00:00Z","value":1},{"at":"2026-08-16T00:00:00Z","value":1},{"at":"2026-08-16T06:00:00Z","value":1},{"at":"2026-08-16T12:00:00Z","value":1},{"at":"2026-08-16T18:00:00Z","value":1},{"at":"2026-08-17T00:00:00Z","value":1},{"at":"2026-08-17T06:00:00Z","value":1},{"at":"2026-08-17T12:00:00Z","value":1},{"at":"2026-08-17T18:00:00Z","value":1},{"at":"2026-08-18T00:00:00Z","value":1},{"at":"2026-08-18T06:00:00Z","value":1},{"at":"2026-08-18T12:00:00Z","value":1},{"at":"2026-08-18T18:00:00Z","value":1},{"at":"2026-08-19T00:00:00Z","value":1},{"at":"2026-08-19T06:00:00Z","value":1},{"at":"2026-08-19T12:00:00Z","value":1},{"at":"2026-08-19T18:00:00Z","value":1},{"at":"2026-08-20T00:00:00Z","value":1},{"at":"2026-08-20T06:00:00Z","value":1},{"at":"2026-08-20T12:00:00Z","value":1},{"at":"2026-08-20T18:00:00Z","value":1},{"at":"2026-08-21T00:00:00Z","value":1},{"at":"2026-08-21T06:00:00Z","value":1},{"at":"2026-08-21T12:00:00Z","value":1},{"at":"2026-08-21T18:00:00Z","value":1},{"at":"2026-08-22T00:00:00Z","value":1},{"at":"2026-08-22T06:00:00Z","value":1},{"at":"2026-08-22T12:00:00Z","value":1},{"at":"2026-08-22T18:00:00Z","value":1},{"at":"2026-08-23T00:00:00Z","value":1},{"at":"2026-08-23T06:00:00Z","value":1},{"at":"2026-08-23T12:00:00Z","value":1},{"at":"2026-08-23T18:00:00Z","value":1},{"at":"2026-08-24T00:00:00Z","value":1},{"at":"2026-08-24T06:00:00Z","value":1},{"at":"2026-08-24T12:00:00Z","value":1},{"at":"2026-08-24T18:00:00Z","value":1},{"at":"2026-08-25T00:00:00Z","value":1},{"at":"2026-08-25T06:00:00Z","value":1},{"at":"2026-08-25T12:00:00Z","value":1},{"at":"2026-08-25T18:00:00Z","value":1},{"at":"2026-08-26T00:00:00Z","value":1},{"at":"2026-08-26T06:00:00Z","value":1},{"at":"2026-08-26T12:00:00Z","value":1}],"db-pg-replica/DatabaseConnections":[{"at":"2026-08-12T18:00:00Z","value":0},{"at":"2026-08-13T00:00:00Z","value":0},{"at":"2026-08-13T06:00:00Z","value":0},{"at":"2026-08-13T12:00:00Z","value":0},{"at":"2026-08-13T18:00:00Z","value":0},{"at":"2026-08-14T00:00:00Z","value":0},{"at":"2026-08-14T06:00:00Z","value":0},{"at":"2026-08-14T12:00:00Z","value":0},{"at":"2026-08-14T18:00:00Z","value":0},{"at":"2026-08-15T00:00:00Z","value":0},{"at":"2026-08-15T06:00:00Z","value":0},{"at":"2026-08-15T12:00:00Z","value":0},{"at":"2026-08-15T18:00:00Z","value":0},{"at":"2026-08-16T00:00:00Z","value":0},{"at":"2026-08-16T06:00:00Z","value":0},{"at":"2026-08-16T12:00:00Z","value":0},{"at":"2026-08-16T18:00:00Z","value":0},{"at":"2026-08-17T00:00:00Z","value":0},{"at":"2026-08-17T06:00:00Z","value":0},{"at":"2026-08-17T12:00:00Z","value":0},{"at":"2026-08-17T18:00:00Z","value":0},{"at":"2026-08-18T00:00:00Z","value":0},{"at":"2026-08-18T06:00:00Z","value":0},{"at":"2026-08-18T12:00:00Z","value":0},{"at":"2026-08-18T18:00:00Z","value":0},{"at":"2026-08-19T00:00:00Z","value":0},{"at":"2026-08-19T06:00:00Z","value":0},{"at":"2026-08-19T12:00:00Z","value":0},{"at":"2026-08-19T18:00:00Z","value":0},{"at":"2026-08-20T00:00:00Z","value":0},{"at":"2026-08-20T06:00:00Z","value":0},{"at":"2026-08-20T12:00:00Z","value":0},{"at":"2026-08-20T18:00:00Z","value":0},{"at":"2026-08-21T00:00:00Z","value":0},{"at":"2026-08-21T06:00:00Z","value":0},{"at":"2026-08-21T12:00:00Z","value":0},{"at":"2026-08-21T18:00:00Z","value":0},{"at":"2026-08-22T00:00:00Z","value":0},{"at":"2026-08-22T06:00:00Z","value":0},{"at":"2026-08-22T12:00:00Z","value":0},{"at":"2026-08-22T18:00:00Z","value":0},{"at":"2026-08-23T00:00:00Z","value":0},{"at":"2026-08-23T06:00:00Z","value":0},{"at":"2026-08-23T12:00:00Z","value":0},{"at":"2026-08-23T18:00:00Z","value":0},{"at":"2026-08-24T00:00:00Z","value":0},{"at":"2026-08-24T06:00:00Z","value":0},{"at":"2026-08-24T12:00:00Z","value":0},{"at":"2026-08-24T18:00:00Z","value":0},{"at":"2026-08-25T00:00:00Z","value":0},{"at":"2026-08-25T06:00:00Z","value":0},{"at":"2026-08-25T12:00:00Z","value":0},{"at":"2026-08-25T18:00:00Z","value":0},{"at":"2026-08-26T00:00:00Z","value":0},{"at":"2026-08-26T06:00:00Z","value":0},{"at":"2026-08-26T12:00:00Z","value":0}],"db-pg-replica/FreeStorageSpace":[{"at":"2026-08-12T18:00:00Z","value":450971566080},{"at":"2026-08-13T00:00:00Z","value":450971566080},{"at":"2026-08-13T06:00:00Z","value":450971566080},{"at":"2026-08-13T12:00:00Z","value":450971566080},{"at":"2026-08-13T18:00:00Z","value":450971566080},{"at":"2026-08-14T00:00:00Z","value":450971566080},{"at":"2026-08-14T06:00:00Z","value":450971566080},{"at":"2026-08-14T12:00:00Z","value":450971566080},{"at":"2026-08-14T18:00:00Z","value":450971566080},{"at":"2026-08-15T00:00:00Z","value":450971566080},{"at":"2026-08-15T06:00:00Z","value":450971566080},{"at":"2026-08-15T12:00:00Z","value":450971566080},{"at":"2026-08-15T18:00:00Z","value":450971566080},{"at":"2026-08-16T00:00:00Z","value":450971566080},{"at":"2026-08-16T06:00:00Z","value":450971566080},{"at":"2026-08-16T12:00:00Z","value":450971566080},{"at":"2026-08-16T18:00:00Z","value":450971566080},{"at":"2026-08-17T00:00:00Z","value":450971566080},{"at":"2026-08-17T06:00:00Z","value":450971566080},{"at":"2026-08-17T12:00:00Z","value":450971566080},{"at":"2026-08-17T18:00:00Z","value":450971566080},{"at":"2026-08-18T00:00:00Z","value":450971566080},{"at":"2026-08-18T06:00:00Z","value":450971566080},{"at":"2026-08-18T12:00:00Z","value":450971566080},{"at":"2026-08-18T18:00:00Z","value":450971566080},{"at":"2026-08-19T00:00:00Z","value":450971566080},{"at":"2026-08-19T06:00:00Z","value":450971566080},{"at":"2026-08-19T12:00:00Z","value":450971566080},{"at":"2026-08-19T18:00:00Z","value":450971566080},{"at":"2026-08-20T00:00:00Z","value":450971566080},{"at":"2026-08-20T06:00:00Z","value":450971566080},{"at":"2026-08-20T12:00:00Z","value":450971566080},{"at":"2026-08-20T18:00:00Z","value":450971566080},{"at":"2026-08-21T00:00:00Z","value":450971566080},{"at":"2026-08-21T06:00:00Z","value":450971566080},{"at":"2026-08-21T12:00:00Z","value":450971566080},{"at":"2026-08-21T18:00:00Z","value":450971566080},{"at":"2026-08-22T00:00:00Z","value":450971566080},{"at":"2026-08-22T06:00:00Z","value":450971566080},{"at":"2026-08-22T12:00:00Z","value":450971566080},{"at":"2026-08-22T18:00:00Z","value":450971566080},{"at":"2026-08-23T00:00:00Z","value":450971566080},{"at":"2026-08-23T06:00:00Z","value":450971566080},{"at":"2026-08-23T12:00:00Z","value":450971566080},{"at":"2026-08-23T18:00:00Z","value":450971566080},{"at":"2026-08-24T00:00:00Z","value":450971566080},{"at":"2026-08-24T06:00:00Z","value":450971566080},{"at":"2026-08-24T12:00:00Z","value":450971566080},{"at":"2026-08-24T18:00:00Z","value":450971566080},{"at":"2026-08-25T00:00:00Z","value":450971566080},{"at":"2026-08-25T06:00:00Z","value":450971566080},{"at":"2026-08-25T12:00:00Z","value":450971566080},{"at":"2026-08-25T18:00:00Z","value":450971566080},{"at":"2026-08-26T00:00:00Z","value":450971566080},{"at":"2026-08-26T06:00:00Z","value":450971566080},{"at":"2026-08-26T12:00:00Z","value":450971566080}],"db-pg-replica/FreeableMemory":[{"at":"2026-08-12T18:00:00Z","value":11811160064},{"at":"2026-08-13T00:00:00Z","value":11811160064},{"at":"2026-08-13T06:00:00Z","value":11811160064},{"at":"2026-08-13T12:00:00Z","value":11811160064},{"at":"2026-08-13T18:00:00Z","value":11811160064},{"at":"2026-08-14T00:00:00Z","value":11811160064},{"at":"2026-08-14T06:00:00Z","value":11811160064},{"at":"2026-08-14T12:00:00Z","value":11811160064},{"at":"2026-08-14T18:00:00Z","value":11811160064},{"at":"2026-08-15T00:00:00Z","value":11811160064},{"at":"2026-08-15T06:00:00Z","value":11811160064},{"at":"2026-08-15T12:00:00Z","value":11811160064},{"at":"2026-08-15T18:00:00Z","value":11811160064},{"at":"2026-08-16T00:00:00Z","value":11811160064},{"at":"2026-08-16T06:00:00Z","value":11811160064},{"at":"2026-08-16T12:00:00Z","value":11811160064},{"at":"2026-08-16T18:00:00Z","value":11811160064},{"at":"2026-08-17T00:00:00Z","value":11811160064},{"at":"2026-08-17T06:00:00Z","value":11811160064},{"at":"2026-08-17T12:00:00Z","value":11811160064},{"at":"2026-08-17T18:00:00Z","value":11811160064},{"at":"2026-08-18T00:00:00Z","value":11811160064},{"at":"2026-08-18T06:00:00Z","value":11811160064},{"at":"2026-08-18T12:00:00Z","value":11811160064},{"at":"2026-08-18T18:00:00Z","value":11811160064},{"at":"2026-08-19T00:00:00Z","value":11811160064},{"at":"2026-08-19T06:00:00Z","value":11811160064},{"at":"2026-08-19T12:00:00Z","value":11811160064},{"at":"2026-08-19T18:00:00Z","value":11811160064},{"at":"2026-08-20T00:00:00Z","value":11811160064},{"at":"2026-08-20T06:00:00Z","value":11811160064},{"at":"2026-08-20T12:00:00Z","value":11811160064},{"at":"2026-08-20T18:00:00Z","value":11811160064},{"at":"2026-08-21T00:00:00Z","value":11811160064},{"at":"2026-08-21T06:00:00Z","value":11811160064},{"at":"2026-08-21T12:00:00Z","value":11811160064},{"at":"2026-08-21T18:00:00Z","value":11811160064},{"at":"2026-08-22T00:00:00Z","value":11811160064},{"at":"2026-08-22T06:00:00Z","value":11811160064},{"at":"2026-08-22T12:00:00Z","value":11811160064},{"at":"2026-08-22T18:00:00Z","value":11811160064},{"at":"2026-08-23T00:00:00Z","value":11811160064},{"at":"2026-08-23T06:00:00Z","value":11811160064},{"at":"2026-08-23T12:00:00Z","value":11811160064},{"at":"2026-08-23T18:00:00Z","value":11811160064},{"at":"2026-08-24T00:00:00Z","value":11811160064},{"at":"2026-08-24T06:00:00Z","value":11811160064},{"at":"2026-08-24T12:00:00Z","value":11811160064},{"at":"2026-08-24T18:00:00Z","value":11811160064},{"at":"2026-08-25T00:00:00Z","value":11811160064},{"at":"2026-08-25T06:00:00Z","value":11811160064},{"at":"2026-08-25T12:00:00Z","value":11811160064},{"at":"2026-08-25T18:00:00Z","value":11811160064},{"at":"2026-08-26T00:00:00Z","value":11811160064},{"at":"2026-08-26T06:00:00Z","value":11811160064},{"at":"2026-08-26T12:00:00Z","value":11811160064}]},"reservations":[{"reservedDBInstanceId":"ri-rds-1","dbInstanceClass":"db.r6i.xlarge","dbInstanceCount":1,"productDescription":"postgresql","offeringType":"All Upfront","state":"active","fixedPrice":2800,"duration":31536000,"startTime":"2026-02-27T12:00:00Z"}],"pageSize":3} diff --git a/pkg/domain/composite_test.go b/pkg/domain/composite_test.go index 43e626e..e5b1c35 100644 --- a/pkg/domain/composite_test.go +++ b/pkg/domain/composite_test.go @@ -156,10 +156,10 @@ func TestEveryRegisteredDomainsKindIsInTheClosedSet(t *testing.T) { } } // And the negative: an unknown kind is refused, so the set stays closed. - if err := r.Register(&partDomain{kind: "rds"}); err == nil { + if err := r.Register(&partDomain{kind: "quantum-annealer"}); err == nil { t.Error("a domain with an unknown kind was registered") } - if _, err := NewComposite("rds", Part{Name: "x", Domain: &partDomain{kind: "rds"}, + if _, err := NewComposite("quantum-annealer", Part{Name: "x", Domain: &partDomain{kind: "quantum-annealer"}, Owns: func(Recommendation) bool { return true }}); err == nil { t.Error("a composite for an unknown kind was built") } diff --git a/pkg/domain/domain.go b/pkg/domain/domain.go index 9623370..c8b0d32 100644 --- a/pkg/domain/domain.go +++ b/pkg/domain/domain.go @@ -44,10 +44,16 @@ const ( EC2 Kind = "ec2" // plain EC2, non-Kubernetes ECSFargate Kind = "ecs-fargate" Lambda Kind = "lambda" + // RDS is managed relational database: a billable domain of its own, not a + // flavour of EC2. pkg/rds declares `const Kind = domain.Kind("rds")` and + // could not be registered until this line existed; pkg/rds/FINDINGS.md §6.1 + // specifies the change and pkg/rds's TestKindIsHonestAboutRegistration + // asserts the property that holds either side of it. + RDS Kind = "rds" ) // kinds is the closed set of known domains, in canonical (sorted) order. -var kinds = []Kind{EC2, ECSFargate, K8sFargate, K8sNodes, Lambda} +var kinds = []Kind{EC2, ECSFargate, K8sFargate, K8sNodes, Lambda, RDS} // Kinds returns a copy of the known domain kinds in canonical order. func Kinds() []Kind { diff --git a/pkg/domain/domain_test.go b/pkg/domain/domain_test.go index dc75577..3dd18b5 100644 --- a/pkg/domain/domain_test.go +++ b/pkg/domain/domain_test.go @@ -23,7 +23,7 @@ func TestKindsAreClosedAndSorted(t *testing.T) { t.Errorf("%q reported invalid but is listed", k) } } - if Kind("k8s-fargate ").Valid() || Kind("").Valid() || Kind("rds").Valid() { + if Kind("k8s-fargate ").Valid() || Kind("").Valid() || Kind("quantum-annealer").Valid() { t.Error("unknown kinds must not validate") } // The returned slice is a copy: mutating it cannot corrupt the table. diff --git a/pkg/domain/hostilerds_test.go b/pkg/domain/hostilerds_test.go new file mode 100644 index 0000000..6059fb1 --- /dev/null +++ b/pkg/domain/hostilerds_test.go @@ -0,0 +1,240 @@ +package domain + +import ( + "context" + "errors" + "testing" + "time" + + "github.com/agenticode/kilter/pkg/pricing/commit" +) + +// The threat model, restated for the domain that has the strongest reason to +// be trusted and must be trusted least. +// +// pkg/rds is structurally read-only: no mutating API appears anywhere in the +// package, Health is unconditionally report-only and PlanSteps refuses +// unconditionally. None of that is why an RDS step cannot run. A [Domain] is +// ordinary Go code, and the value of `rds` as a registry key is now a public +// constant that anything in-process can claim. What stops an impostor is the +// core: Health is checked BEFORE the domain is consulted, steps are validated +// on the way OUT, and execution routes through an actuator table only cmd/ can +// write to. +// +// impostorDomain is that impostor. Every field is a separate lie so each can +// be tested alone, and it lies through the two seams the RDS wiring newly +// depends on — [Refuser] and the usage-line projection — which the existing +// hostile-domain tests do not exercise. +type impostorDomain struct { + kind Kind + // claimsActuatable makes Health report ReportOnly:false on a domain with + // no collector, no credentials and no actuator. + claimsActuatable bool + // steps is handed back unconditionally, including for recommendations it + // was never given and targets in domains it does not own. + steps []Step + // recs are fabricated recommendations. netAboveGross populates the two + // savings fields DIRECTLY, bypassing SetSavings — the one clamp that makes + // Net ≤ Gross an invariant rather than a convention. + recs []Recommendation + // refusals are fabricated refusals, optionally attributed to another + // domain's targets. + refusals []Refusal + // lines are fabricated account-wide usage lines. + lines []commit.UsageLine + + planCalls int +} + +func (i *impostorDomain) Kind() Kind { return i.kind } +func (i *impostorDomain) Learn(*Snapshot) error { return nil } +func (i *impostorDomain) Recommend(time.Time, Netter) []Recommendation { + out := make([]Recommendation, len(i.recs)) + copy(out, i.recs) + return out +} +func (i *impostorDomain) PlanSteps([]Recommendation, Guard) ([]Step, error) { + i.planCalls++ + return i.steps, nil +} +func (i *impostorDomain) Health(time.Time) Health { + return Health{Kind: i.kind, Ready: true, ReportOnly: !i.claimsActuatable, Targets: 1} +} +func (i *impostorDomain) Checkpoint() ([]byte, error) { return nil, nil } +func (i *impostorDomain) Restore([]byte) error { return nil } + +// Refusals implements [Refuser]. +func (i *impostorDomain) Refusals(time.Time, Netter) []Refusal { + out := make([]Refusal, len(i.refusals)) + copy(out, i.refusals) + return out +} + +// UsageLines is the shape cmd/ reaches for when it splices the account-wide +// commitment baseline. +func (i *impostorDomain) UsageLines(time.Time, Netter) []commit.UsageLine { + out := make([]commit.UsageLine, len(i.lines)) + copy(out, i.lines) + return out +} + +var impostorNow = time.Date(2026, 8, 26, 12, 0, 0, 0, time.UTC) + +// TestAHostileRDSDomainCannotActuateOrBorrowAnotherDomainsActuator. +// +// Three independent claims, in increasing order of severity: +// +// 1. It says it is not report-only. The core does not grant actuation on a +// domain's own say-so — CanActuate reads the table, not the Health. +// 2. It hands back a well-formed step for its OWN kind. There is no actuator +// for `rds` and there cannot be one, so Execute and Revert both refuse. +// 3. It hands back a step labelled `ec2`, where a real actuator IS wired. +// This is the hole j1-wire closed: steps are routed for execution by +// Step.Target.Domain, so an unvalidated output would let any domain borrow +// any other domain's actuator and its credentials. +func TestAHostileRDSDomainCannotActuateOrBorrowAnotherDomainsActuator(t *testing.T) { + victim := &recordingActuator{kind: EC2} + imp := &impostorDomain{kind: RDS, claimsActuatable: true} + + r := NewRegistry() + if err := r.Register(imp); err != nil { + t.Fatal(err) + } + if err := r.Register(ready(EC2)); err != nil { + t.Fatal(err) + } + if err := r.RegisterActuator(victim); err != nil { + t.Fatal(err) + } + + // 1. Claiming actuatability grants nothing. + if r.CanActuate(RDS) { + t.Fatal("an rds domain granted itself actuation capability") + } + + // 2. Its own step goes nowhere. + own := stepFor(RDS, "db-prod-1") + if err := r.Execute(context.Background(), Guard{Now: impostorNow}, own); !errors.Is(err, ErrReportOnly) { + t.Errorf("Execute(rds step) = %v, want ErrReportOnly", err) + } + if err := r.Revert(context.Background(), Guard{Now: impostorNow}, own); !errors.Is(err, ErrReportOnly) { + t.Errorf("Revert(rds step) = %v, want ErrReportOnly", err) + } + + // 3. A step labelled as EC2's is refused on the way out of the domain, so + // it never becomes something Execute could route. + rec := validRec() + rec.Target.Domain = RDS + imp.steps = []Step{stepFor(EC2, "i-victim")} + steps, err := r.PlanSteps(RDS, []Recommendation{rec}, Guard{Now: impostorNow}) + if !errors.Is(err, ErrWrongDomain) { + t.Fatalf("PlanSteps = (%v, %v), want ErrWrongDomain", steps, err) + } + if len(steps) != 0 { + t.Fatal("a cross-domain step escaped the core") + } + p := r.BuildPlan(RDS, []Recommendation{rec}, Guard{Now: impostorNow}) + if p.RefusalCode != RefuseDomainError { + t.Errorf("BuildPlan refusal = %q, want %q", p.RefusalCode, RefuseDomainError) + } + if len(victim.executed) != 0 || len(victim.reverted) != 0 { + t.Fatal("the ec2 actuator was reached by a domain that does not own it") + } + if imp.planCalls == 0 { + t.Fatal("the test never reached the domain's PlanSteps; it proves nothing") + } +} + +// TestAnHonestRDSDomainIsAlsoRefused is the control, and it is the point of +// the whole arrangement: the outcome does not depend on the domain's manners. +// +// pkg/rds's real Health is report-only, so the core refuses one step EARLIER — +// before PlanSteps is called at all. Both a liar and an honest domain end at +// the same wall; only the number of walls they hit differs. +func TestAnHonestRDSDomainIsAlsoRefused(t *testing.T) { + honest := &impostorDomain{kind: RDS, claimsActuatable: false} + r := NewRegistry() + if err := r.Register(honest); err != nil { + t.Fatal(err) + } + rec := validRec() + rec.Target.Domain = RDS + if _, err := r.PlanSteps(RDS, []Recommendation{rec}, Guard{Now: impostorNow}); !errors.Is(err, ErrReportOnly) { + t.Fatalf("PlanSteps = %v, want ErrReportOnly", err) + } + if honest.planCalls != 0 { + t.Error("a report-only domain's PlanSteps was called; the core deferred to the domain") + } +} + +// TestAHostileDomainCannotAttributeFindingsToAnotherDomain. +// +// The RDS wiring is the first that depends on [Refuser] for its entire output, +// which makes the seam worth attacking. A domain that returned refusals +// pointing at another domain's targets would put words in that domain's mouth +// — a `not-gp2` line under `ec2` that pkg/ebs never said, in a report a human +// is asked to act on. The registry re-stamps the producing kind onto every +// refusal for the same reason Recommend re-stamps it onto every +// recommendation. +func TestAHostileDomainCannotAttributeFindingsToAnotherDomain(t *testing.T) { + imp := &impostorDomain{kind: RDS, refusals: []Refusal{ + {Target: TargetRef{Domain: EC2, Scope: "acct/us-east-1", ID: "i-victim"}, + Code: "not-gp2", Reason: "a finding pkg/ebs never made"}, + {Target: TargetRef{Domain: Lambda, Scope: "acct/us-east-1", ID: "fn-victim"}, + Code: "single-memory-point", Reason: "nor did pkg/lambda"}, + }} + r := NewRegistry() + if err := r.Register(imp); err != nil { + t.Fatal(err) + } + if err := r.Register(ready(EC2)); err != nil { + t.Fatal(err) + } + + got := r.Refusals(RDS, impostorNow, nil) + if len(got) != 2 { + t.Fatalf("got %d refusals, want 2", len(got)) + } + for _, ref := range got { + if ref.Target.Domain != RDS { + t.Errorf("refusal for %s is attributed to %q; the producing domain must own its output", + ref.Target.ID, ref.Target.Domain) + } + } + // And the aggregate agrees: EC2's row counts none of them. + rep := Summarize(impostorNow, r, nil) + ec2Row, ok := rep.For(EC2) + if !ok { + t.Fatal("ec2 is missing from the report") + } + if ec2Row.Refused != 0 { + t.Errorf("ec2 was charged %d refusals it never made", ec2Row.Refused) + } +} + +// TestAFabricatedSavingIsCaughtByReportValidate. +// +// Recommendation.SetSavings clamps Net ≤ Gross, which is what makes the +// invariant mechanical — but a domain can assign the fields directly and skip +// the clamp. Nothing in Summarize re-checks it, by design: the aggregate is a +// projection, and Report.Validate is the gate. The CLI calls Validate before +// printing and fails loudly rather than print a number somebody might put in a +// business case, so this asserts the gate rather than the projection. +func TestAFabricatedSavingIsCaughtByReportValidate(t *testing.T) { + rec := validRec() + rec.Target.Domain = RDS + // The clamp, bypassed. + rec.GrossSavingsMonthlyUSD = 1 + rec.NetSavingsMonthlyUSD = 100_000 + + imp := &impostorDomain{kind: RDS, recs: []Recommendation{rec}} + r := NewRegistry() + if err := r.Register(imp); err != nil { + t.Fatal(err) + } + rep := Summarize(impostorNow, r, nil) + if err := rep.Validate(); err == nil { + t.Fatalf("a report claiming net $%v on gross $%v validated", + rec.NetSavingsMonthlyUSD, rec.GrossSavingsMonthlyUSD) + } +} diff --git a/pkg/domain/rds/rds.go b/pkg/domain/rds/rds.go new file mode 100644 index 0000000..755243b --- /dev/null +++ b/pkg/domain/rds/rds.go @@ -0,0 +1,120 @@ +// Package rds wires pkg/rds's read-only observation domain into the seam, and +// contributes the one thing that package deliberately left to the wiring: the +// account-wide commitment usage baseline. +// +// pkg/rds is the first domain whose entire product is refusals. It proposes +// nothing — the DB instance class is unrepresentable in [rds.Proposal], not +// merely forbidden — so [rds.Domain.Recommend] returns an empty slice and +// [rds.Domain.Refusals] returns everything. The adapter therefore adds no +// recommendation plumbing; the seam already renders refusals beside +// recommendations, which is exactly what this domain needed to exist. +// +// # What the adapter adds, and why it could not live in pkg/rds +// +// One method: [Domain.UsageLines]. pkg/rds ships [rds.UsageLines] as a pure +// function over one instance because the *sizer* needs it to build a +// before/after pair, and it stops there on purpose — a domain knows only its +// own targets and must not be tempted to construct an account-wide view +// (pkg/domain/ledger.go's argument for [domain.Netter]). Projecting a whole +// report into baseline lines is the brain's job, and the brain reaches for it +// through the `UsageLines(now, ledger)` shape cmd/ already uses for pkg/ec2 +// and pkg/ebs. +// +// # Report-only is not this package's promise to keep +// +// pkg/rds's Health is unconditionally report-only and its PlanSteps refuses +// unconditionally, and neither fact is why nothing can be actuated here. The +// core refuses first: [domain.Registry.PlanSteps] checks Health before the +// domain is consulted, and [domain.Registry.Execute] routes through an +// actuator table only cmd/ can write to, which has no `rds` row and cannot +// grow one — there is no mutating RDS API anywhere in the tree. +// TestAHostileRDSDomainCannotActuateOrBorrowAnotherDomainsActuator in +// pkg/domain asserts that against a domain built to lie about all three. +package rds + +import ( + "fmt" + "time" + + "github.com/agenticode/kilter/pkg/domain" + "github.com/agenticode/kilter/pkg/pricing/commit" + krds "github.com/agenticode/kilter/pkg/rds" +) + +// Kind is the compute domain this adapter serves. It is the same value +// pkg/rds declares, and it became registrable when pkg/domain's closed set +// grew the row (pkg/rds/FINDINGS.md §6.1). +const Kind = domain.RDS + +// Config wires the domain. +type Config struct { + // Scope is the accountID/region this collector covers. + Scope string + // Region labels commitment usage lines. + Region string + // Rates prices instance-hours and storage. The zero value means + // [rds.DefaultRates], every row of which is `unverified` and therefore + // able to size a reported fact and unable to become a claimed saving. + Rates krds.RateCard +} + +// Domain is pkg/rds's read-only domain plus the account-wide usage projection. +type Domain struct { + *krds.Domain +} + +// New builds the domain. +func New(cfg Config) (*Domain, error) { + sc := krds.DefaultConfig() + sc.Scope, sc.Region = cfg.Scope, cfg.Region + if len(cfg.Rates.Classes) > 0 { + sc.Rates = cfg.Rates + } + d, err := krds.NewDomain(sc) + if err != nil { + return nil, fmt.Errorf("domain/rds: %w", err) + } + return &Domain{Domain: d}, nil +} + +// UsageLines projects every priced DB instance into the account-wide +// commitment baseline, as `cmd/`'s two-pass ledger build expects. +// +// # Which instances contribute, and why that is the right set +// +// Exactly the assessments pkg/rds could price. An EXCLUDED instance — Aurora, +// a Multi-AZ DB cluster member, an unknown engine, an unreadable topology, +// `kilter.dev/mode=off` — never reaches the price step in +// `Sizer.assessTarget`, so its CostKnown is false and it contributes nothing. +// That is the honest direction: a baseline line for an instance nobody could +// price would carry a rate of zero, and a zero-rate line in the baseline makes +// a Reserved DB Instance look like it is absorbing usage that costs nothing, +// which OVERSTATES absorption and therefore overstates every other domain's +// saving. Under-claiming is the only safe way to be wrong here. +// +// The ledger argument is accepted to satisfy the shape `cmd/` calls through +// and is deliberately unused: the baseline is what the account costs today at +// on-demand rates, which is what a commitment is applied *to*. Netting it +// against the commitments while computing it would be circular. +func (d *Domain) UsageLines(now time.Time, _ domain.Netter) []commit.UsageLine { + rep := d.Report(now, nil) + if rep == nil { + return nil + } + out := make([]commit.UsageLine, 0, len(rep.Assessments)) + for _, a := range rep.Assessments { + if !a.CostKnown { + continue + } + // Same call the sizer's own before/after pair is built from + // (pkg/rds/sizer.go), with the DEPLOYMENT-ADJUSTED hourly rate — so a + // Multi-AZ instance's line costs twice a Single-AZ one's exactly as it + // consumes twice the normalized units. The two halves of trap 10 stay + // in step because they come from one multiplier. + out = append(out, krds.UsageLines(a.Target, a.Instance, a.Engine, a.Deployment, a.CurrentHourlyUSD)...) + } + if len(out) == 0 { + return nil + } + return out +} diff --git a/pkg/domain/rds/rds_test.go b/pkg/domain/rds/rds_test.go new file mode 100644 index 0000000..1d93cfa --- /dev/null +++ b/pkg/domain/rds/rds_test.go @@ -0,0 +1,170 @@ +package rds + +import ( + "testing" + "time" + + "github.com/agenticode/kilter/pkg/domain" + "github.com/agenticode/kilter/pkg/pricing/commit" + krds "github.com/agenticode/kilter/pkg/rds" +) + +var testNow = time.Date(2026, 8, 26, 12, 0, 0, 0, time.UTC) + +func newTestDomain(t *testing.T) *Domain { + t.Helper() + d, err := New(Config{Scope: "000000000000/us-east-1", Region: "us-east-1"}) + if err != nil { + t.Fatal(err) + } + return d +} + +func snapshotWith(insts ...krds.DBInstance) *krds.Snapshot { + w := krds.Window{Start: testNow.Add(-14 * 24 * time.Hour), End: testNow} + s := &krds.Snapshot{ + Domain: krds.Kind, Scope: "000000000000/us-east-1", Region: "us-east-1", + Timestamp: testNow, Window: w, + } + for _, in := range insts { + s.Targets = append(s.Targets, krds.Target{ + Ref: domain.TargetRef{Domain: domain.RDS, Scope: s.Scope, ID: in.ARN, Name: in.Identifier}, + Instance: in, + }) + } + krds.SortTargets(s.Targets) + return s +} + +func instance(id, class, engine string, multiAZ bool) krds.DBInstance { + return krds.DBInstance{ + ARN: "arn:aws:rds:us-east-1:000000000000:db:" + id, Identifier: id, + Class: class, Engine: engine, LicenseModel: krds.LicenseGPL, + Status: krds.StatusAvailable, Region: "us-east-1", MultiAZ: multiAZ, + AllocatedStorageGiB: 100, StorageType: krds.StorageGP2, + } +} + +// TestRegistersAndStaysReportOnly: the kind is registrable now, and +// registering it grants nothing. +func TestRegistersAndStaysReportOnly(t *testing.T) { + d := newTestDomain(t) + if d.Kind() != domain.RDS { + t.Fatalf("Kind() = %q, want %q", d.Kind(), domain.RDS) + } + reg := domain.NewRegistry() + if err := reg.Register(d); err != nil { + t.Fatalf("Register: %v", err) + } + if reg.CanActuate(domain.RDS) { + t.Error("registering the domain wired an actuator") + } + if h := d.Health(testNow); !h.ReportOnly { + t.Error("the rds domain is not report-only") + } + if _, err := reg.PlanSteps(domain.RDS, nil, domain.Guard{Now: testNow}); err == nil { + t.Error("the core planned steps for a report-only domain") + } + if _, ok := any(d).(domain.Refuser); !ok { + t.Error("the adapter does not implement Refuser; the refusals ARE this domain's output") + } +} + +// TestUsageLinesCarryTheDeploymentMultiplier: the Multi-AZ line costs exactly +// twice the Single-AZ line for the same class, because both come from the same +// multiplier that the reservation arithmetic uses. +func TestUsageLinesCarryTheDeploymentMultiplier(t *testing.T) { + d := newTestDomain(t) + if err := d.Observe(snapshotWith( + instance("single", "db.r6i.xlarge", "postgres", false), + instance("multi", "db.r6i.xlarge", "postgres", true), + )); err != nil { + t.Fatal(err) + } + byID := map[string]commit.UsageLine{} + for _, l := range d.UsageLines(testNow, nil) { + byID[l.ID] = l + } + if len(byID) != 2 { + t.Fatalf("got %d usage lines, want 2: %v", len(byID), byID) + } + var single, multi commit.UsageLine + for id, l := range byID { + if l.Kind != commit.KindRDS { + t.Errorf("%s: kind %q, want %q", id, l.Kind, commit.KindRDS) + } + if l.Deployment == commit.RDSMultiAZInstance { + multi = l + } else { + single = l + } + } + if single.ODRate <= 0 || multi.ODRate != 2*single.ODRate { + t.Errorf("multi $%v, single $%v — want exactly 2×", multi.ODRate, single.ODRate) + } +} + +// TestUnpricedInstancesNeverEnterTheBaseline. +// +// A baseline line for an instance nobody could price carries a rate of zero, +// and a zero-rate line makes a Reserved DB Instance look like it is absorbing +// usage that costs nothing — which overstates absorption and therefore +// overstates every OTHER domain's saving. Under-claiming is the only safe way +// to be wrong here. +func TestUnpricedInstancesNeverEnterTheBaseline(t *testing.T) { + aurora := instance("aurora-1", "db.r6i.large", "aurora-postgresql", false) + aurora.ClusterID = "aurora-prod" + unknownEngine := instance("weird-1", "db.r6i.large", "cockroach", false) + unpricedClass := instance("odd-1", "db.zz9.plural-z-alpha", "postgres", false) + licensed := instance("mssql-1", "db.r6i.xlarge", "sqlserver-ee", false) + licensed.LicenseModel = krds.LicenseIncluded + priced := instance("ok-1", "db.r6i.large", "postgres", false) + + d := newTestDomain(t) + if err := d.Observe(snapshotWith(aurora, unknownEngine, unpricedClass, licensed, priced)); err != nil { + t.Fatal(err) + } + lines := d.UsageLines(testNow, nil) + if len(lines) != 1 { + t.Fatalf("got %d usage lines, want only the priced one: %+v", len(lines), lines) + } + if lines[0].ID != priced.ARN+"/instance" { + t.Errorf("the wrong instance reached the baseline: %s", lines[0].ID) + } + // And every refusal is still reported: excluded is not invisible. + if got := len(d.Refusals(testNow, nil)); got < 4 { + t.Errorf("%d refusals for 5 instances; an excluded instance must still be reported", got) + } +} + +// TestNothingLearnedProducesNoLinesRatherThanZeroes. +func TestNothingLearnedProducesNoLinesRatherThanZeroes(t *testing.T) { + d := newTestDomain(t) + if lines := d.UsageLines(testNow, nil); lines != nil { + t.Errorf("a domain with no snapshot produced %d baseline lines", len(lines)) + } + if recs := d.Recommend(testNow, nil); len(recs) != 0 { + t.Errorf("a domain that proposes nothing produced %d recommendations", len(recs)) + } +} + +// TestOperatorRatesAreLayeredNotReplaced: an operator supplying only the +// licensed rows must not lose the open-source ones. +func TestOperatorRatesAreLayeredNotReplaced(t *testing.T) { + card := krds.DefaultRates() + base, _, ok := card.HourlyUSD("db.r6i.large", krds.ParseEngine("postgres", krds.LicenseGPL), commit.RDSSingleAZ) + if !ok { + t.Fatal("the shipped card cannot price a db.r6i.large PostgreSQL instance") + } + d, err := New(Config{Scope: "s", Region: "us-east-1", Rates: card}) + if err != nil { + t.Fatal(err) + } + if err := d.Observe(snapshotWith(instance("ok-1", "db.r6i.large", "postgres", false))); err != nil { + t.Fatal(err) + } + lines := d.UsageLines(testNow, nil) + if len(lines) != 1 || lines[0].ODRate != base { + t.Errorf("rate %v, want the card's %v", lines, base) + } +} diff --git a/pkg/domain/registry_test.go b/pkg/domain/registry_test.go index 90f136a..a0618e9 100644 --- a/pkg/domain/registry_test.go +++ b/pkg/domain/registry_test.go @@ -109,7 +109,7 @@ func TestRegisterRejectsWiringBugs(t *testing.T) { if err := r.Register(nil); err == nil { t.Error("Register(nil) accepted") } - if err := r.Register(&fakeDomain{kind: "rds"}); err == nil { + if err := r.Register(&fakeDomain{kind: "quantum-annealer"}); err == nil { t.Error("Register accepted an unknown kind") } if err := r.Register(ready(K8sFargate)); err != nil { diff --git a/pkg/domain/report_test.go b/pkg/domain/report_test.go index ee7721d..66652b2 100644 --- a/pkg/domain/report_test.go +++ b/pkg/domain/report_test.go @@ -235,7 +235,7 @@ func TestReportValidateCatchesEachViolation(t *testing.T) { {"total recs disagree", func(r *Report) { r.Totals.Recommendations = 2; r.Totals.Applicable = 2 }}, {"domain count disagrees", func(r *Report) { r.Totals.Domains = 7 }}, {"duplicate domain", func(r *Report) { r.Domains = append(r.Domains, r.Domains[0]); r.Totals.Domains = 2 }}, - {"unknown kind", func(r *Report) { r.Domains[0].Kind = "rds" }}, + {"unknown kind", func(r *Report) { r.Domains[0].Kind = "quantum-annealer" }}, {"invalid recommendation", func(r *Report) { r.Recommendations[0].Evidence = nil }}, {"refusal with no code", func(r *Report) { r.Refusals = []Refusal{{Target: TargetRef{ID: "x"}, Reason: "why"}}