From 2ba5df2cc5e62f6330ff76601e903abeab01f7db Mon Sep 17 00:00:00 2001 From: agenticode <16611333+agenticode@users.noreply.github.com> Date: Wed, 26 Aug 2026 18:28:13 +0900 Subject: [PATCH] refactor(confidence): lift the additive confidence model into pkg/confidence Co-authored-by: kording <74226694+kording@users.noreply.github.com> --- pkg/confidence/FINDINGS.md | 317 ++++++++++++++++++++++++++++ pkg/confidence/confidence.go | 138 ++++++++++++ pkg/confidence/confidence_test.go | 185 ++++++++++++++++ pkg/ec2/confidence_equiv_test.go | 222 +++++++++++++++++++ pkg/ec2/sizer.go | 62 ++---- pkg/lambda/confidence_equiv_test.go | 236 +++++++++++++++++++++ pkg/lambda/sizer.go | 63 ++---- 7 files changed, 1137 insertions(+), 86 deletions(-) create mode 100644 pkg/confidence/FINDINGS.md create mode 100644 pkg/confidence/confidence.go create mode 100644 pkg/confidence/confidence_test.go create mode 100644 pkg/ec2/confidence_equiv_test.go create mode 100644 pkg/lambda/confidence_equiv_test.go diff --git a/pkg/confidence/FINDINGS.md b/pkg/confidence/FINDINGS.md new file mode 100644 index 0000000..7d930ea --- /dev/null +++ b/pkg/confidence/FINDINGS.md @@ -0,0 +1,317 @@ +# `pkg/confidence` — the differential first, the lift second + +`pkg/rds/FINDINGS.md` §7.3 asks for this package, and describes the situation +as *"the three shipped domains each wrote their own"* confidence model. That +premise is wrong in a way that matters, and finding out how was most of this +unit. + +**There are not three copies of one model. There are two models with two users +each.** `pkg/ec2` and `pkg/lambda` share an *additive* model — score starts at +zero and each factor adds `weight × earned`. `pkg/ecs` and +`pkg/domain/fargate` share a *multiplicative* one, `decision.Compose`, which +already lives in a shared package (`pkg/decision`) and reproduces the +Kubernetes recommender's historical formula by construction. + +So the fourth copy §7.3 warns about is real, but it is the third copy of the +*additive* model. That is what this package holds. `pkg/ecs` is not touched and +must not be. + +--- + +## 1. The factor-by-factor differential + +`✅` = genuinely identical. `❌` = genuinely different. + +### 1.1 Model shape + +| | `pkg/ec2` | `pkg/ecs` | `pkg/lambda` | | +|---|---|---|---|---| +| Composition | `Σ wᵢ·eᵢ` | `∏ vᵢ^wᵢ` | `Σ wᵢ·eᵢ` | ❌ ecs | +| Public type | `Confidence{Score, Factors}` | bare `float64` | `Confidence{Score, Factors}` | ❌ ecs | +| Rounding | none | `Round(s·100)/100` | none | ❌ ecs | +| Zero evidence | 0 | 0 | 0 | ✅ | +| Weights sum to | 1 (asserted) | n/a (exponents, all 1) | 1 (asserted) | ❌ ecs | +| Reads a clock | no | **yes** (`now`, for freshness) | no | ❌ ecs | +| `weakestFactor` | yes | **absent** | yes | ❌ ecs | +| Term list retained | yes | **discarded** (`return c.Score`) | yes | ❌ ecs | + +Between `ec2` and `lambda` the shape agrees on every row, and +`ConfidenceFactor` / `Confidence` were byte-identical declarations including +JSON tags. `weakestFactor` was byte-identical in both. + +### 1.2 The accumulator (`add`) + +| earned | `pkg/ec2` | `pkg/lambda` | | +|---|---|---|---| +| `< 0` | → 0 | → 0 | ✅ | +| `> 1` | → 1 | → 1 | ✅ | +| `-Inf` | → 0 | → 0 | ✅ | +| **`NaN`** | **passes through, poisons `Score`** | **→ 0** | ❌ | +| **`+Inf`** | **→ 1** | **→ 0** | ❌ | + +`pkg/ec2` clamps by comparison alone; NaN satisfies neither `< 0` nor `> 1`. +`pkg/lambda` guards with `finite()` first. + +### 1.3 The factors themselves + +| Factor | ec2 | lambda | | +|---|---|---|---| +| `window` — earned | `w.Seconds()/min.Seconds()`, 0 if `min ≤ 0` | identical | ✅ | +| `window` — prose | `…against a %s minimum` | identical format | ✅ | +| `window` — prose rounding | `Round(time.Hour)` | `Round(time.Minute)` | ❌ | +| `window` — weight | 0.20 | 0.15 | ❌ | +| coverage | `sample-coverage`, 0.30, pass-through `obs.Coverage` | `report-coverage`, 0.25, pass-through `obs.ReportCoverage` | ❌ name/weight/prose, ✅ arithmetic | +| everything else | `memory-signal` 0.20, `metric-resolution` 0.15, `burst-evidence` 0.15 | `measured-points` 0.30, `warm-share` 0.15, `memory-headroom` 0.15 | ❌ disjoint | + +--- + +## 2. The rulings + +### 2.1 `pkg/ecs` stays where it is — DOMAIN FACT, not accidental + +Not lifted, and `pkg/ecs/sizer.go` is unmodified. Four independent reasons, any +one of which is sufficient: + +1. **The number would move.** A product is not a sum. ECS also rounds to two + decimals inside `decision.Compose`; the additive model does not round at + all. Every shipped ECS confidence value would change. +2. **The type would move.** `Assessment.Confidence` is `float64` + (`pkg/ecs/sizer.go:187`), not a struct. Changing it violates the + additive-only rule on exported signatures. +3. **ECS's lineage is the recommender, not EC2.** `decision.Compose`'s own doc + says it reproduces the recommender's historical formula "by construction". + `pkg/domain/fargate/recommend.go:423` composes the same way. Moving ECS to + the additive model would break its agreement with the two things it is + actually meant to agree with, to gain agreement with two things it was never + meant to agree with. +4. **ECS has a factor EC2 and Lambda deliberately do not.** `TermFreshness` + takes `now`. The additive model is pure over a closed observation window and + reads no clock; adding freshness to it would make it clock-dependent, which + is a different guarantee. + +§7.3's own wording already knew this: it names the shape to copy as *"from +`pkg/ec2`/`pkg/lambda`"* and does not name `pkg/ecs`. + +### 2.2 The `NaN` / `+Inf` divergence — ACCIDENTAL, and preserved anyway + +`pkg/lambda` is right and `pkg/ec2` is older. A confidence factor that scores +NaN as evidence is a bug in waiting, and `finite()` is the fix. + +**It is still not this unit's fix to make.** Flattening it would change +`pkg/ec2`'s output on non-finite input — which is exactly the failure mode this +unit exists to prevent, and it would be invisible in a diff that reads as a +refactor. So both spellings survive as +[`Confidence.Add`](confidence.go) (guarded) and `Confidence.AddBounded` +(comparison-only), and the equivalence proof pins each domain to the one it +shipped with. + +`pkg/ec2` adopting `Add` is now a one-word diff in one file, and +`TestConfidenceKeepsItsNaNPropagation` is the test that will fail loudly when +someone makes it. That is the whole point: the divergence went from *invisible +and duplicated* to *visible and named*. + +Reachability, for whoever makes that call: `obs.Coverage` is +`math.Min(1, float64(int)/float64(int))` behind an `ExpectedSamples > 0` guard +(`pkg/ec2/sizer.go:520`), so NaN is **[unverified] believed unreachable in +production today**. The proof does not depend on that belief — it pins the +behaviour over the whole `float64` domain either way. + +### 2.3 The `window` prose rounding — DOMAIN FACT + +`Round(time.Hour)` vs `Round(time.Minute)` is not a style difference. An EC2 +minimum window is quoted in days (`DefaultConfig().MinWindow` = 7 days); a +Lambda log window is a slice of hours (24h). + +Note carefully: **at both shipped defaults the two spellings print the same +string**, because 7 days and 24 h are both whole hours. The divergence only +bites when `MinWindow` is set to something that is not — and `MinWindow` is a +caller-settable `Config` field, so that is a supported configuration, not a +hypothetical. Under the EC2 spelling a 6h20m Lambda minimum prints as +`"6h0m0s"`: a wrong number, in a line whose entire job is to tell an operator +how much more window they need. + +This is the case for the dense sweep rather than a handful of realistic +fixtures. A proof built only from default configs would have found the two +spellings interchangeable and flattened them. It survives as the `round` +parameter of `WindowFactor`, never as a constant. + +### 2.4 The weights — DOMAIN FACT + +Both budgets sum to 1 and each domain has a test asserting it, but `window` is +worth 0.20 to EC2 and 0.15 to Lambda, and no other factor name is even shared. +These are claims about what each domain's evidence is worth. They stay in their +domains. This package holds no weights. + +--- + +## 3. What was lifted, and what was left behind + +**Lifted** (`confidence.go`, 138 lines): `Factor`, `Confidence`, `Add`, +`AddBounded`, `WeakestFactor`, `FactorWindow`, `WindowFactor`. + +`pkg/ec2` and `pkg/lambda` now alias the types rather than redeclare them, so +the two reports cannot drift apart in JSON. Net effect on the two domains: +**−47 lines**, and the `add`/`weakestFactor` bodies exist once. + +**Left behind, deliberately:** + +| Not lifted | Why | +|---|---| +| The coverage factor | The arithmetic is a pass-through — a lifted helper would be `func(x) x`. Only the name, weight and prose differ, and those are the domain's claims. Lifting it would move code without removing a decision. | +| `memory-signal`, `metric-resolution`, `burst-evidence`, `measured-points`, `warm-share`, `memory-headroom` | Disjoint. There is no second implementation to deduplicate. | +| The weight constants | §2.4. | +| Any `Score`-gating helper | `MinConfidence` comparison plus a refusal string is three lines and each domain words its refusal differently. | +| `pkg/ecs`'s model | §2.1. | + +`WindowFactor` is the only *judgement* that moved, and it moved because both +domains computed it identically — including the detail below, which is the +reason it was worth moving at all rather than leaving as six duplicated lines. + +**The division is `observed.Seconds() / minimum.Seconds()`, not +`float64(observed) / float64(minimum)`.** These are not the same number. +`Duration.Seconds()` is `float64(d/1e9) + float64(d%1e9)/1e9`, not a single +division, and the two disagree in the last bit. Measured, by mutating the +implementation and running the proof: for a 3h7m13.456789123s window against a +3h20m minimum, `Seconds()/Seconds()` gives `0.9361213990935833` +(`0x3fedf4b4dd462aa2`) and `float64/float64` gives `0.9361213990935834` +(`0x3fedf4b4dd462aa3`). One ULP — and a shipped confidence value. A +reimplementation from the doc comment would have gotten this wrong. + +--- + +## 4. The equivalence proof + +`pkg/ec2/confidence_equiv_test.go` and `pkg/lambda/confidence_equiv_test.go` +each carry the **pre-lift implementation verbatim** — the old struct, the old +`add`, the old `confidence` body, the old `weakestFactor` — and assert the +shipped path reproduces it over a dense input sweep. + +**Compared by raw bits** (`math.Float64bits`), not by tolerance: `Score`, and +for every factor its `Name`, `Weight`, `Earned` and `Why`. A confidence that +moves in the last bit is still a confidence that moved. + +| | inputs swept | combinations | +|---|---|---| +| `pkg/ec2` | 6 minimum windows × 8 windows × 11 coverages × 7 periods × 7 burst classes × memory-blind | **56,448** | +| `pkg/lambda` | 5 minimum windows × 7 windows × 9 coverages × 7 cold shares × 9 point sets × 4 current-indices × warm | **158,760** | + +Coverage inputs include `NaN`, `±Inf`, `-0.0`, `MaxFloat64` and +`0.9999999999999999`; windows include a span whose `Seconds()` is not exactly +representable; burst classes include `""` and an unknown future class; Lambda's +point sets cover every branch of `measured-points` (0/1/2/3+) and of +`memory-headroom` (absent, zero-memory, at-ceiling, margins either side of the +0.25 saturation, and a negative margin). + +**`weakestFactor` is pinned in the same assertion**, on every one of those +combinations — the string, not just the score. It is a reporting surface: an +operator reads it to know what to go measure, so a change in *which* factor is +named is a behaviour change even when the number is unchanged. +`TestWeakestFactorNamesTheLargestLossAndBreaksTiesByOrder` additionally pins +the tie-break (earliest factor wins, via strict `>`), which is what makes the +result a function of call order alone. + +### 4.1 The proof was checked against mutants + +A proof that cannot fail proves nothing. Each of these was applied to the +shipped code, confirmed to fail the proof, and reverted: + +| Mutation | Caught by | +|---|---| +| `AddBounded` → `Add` in `pkg/ec2` (the "harmless cleanup") | `score: legacy NaN != lifted 0.5` | +| `Round(time.Hour)` → `Round(time.Minute)` in `pkg/ec2` | `factor 1 (window) why` | +| `Round(time.Minute)` → `Round(time.Hour)` in `pkg/lambda` | `factor 3 (window) why` | +| `Seconds()/Seconds()` → `float64/float64` in `WindowFactor` | `factor 1 (window) earned`, 1 ULP | +| `WeakestFactor` tie-break `>` → `>=` | `weakestFactor` string, both domains | + +### 4.2 What the proof does *not* cover, stated plainly + +Mutating `pkg/lambda`'s window factor from `Add` to `AddBounded` **does not** +fail the proof. That is not a gap in the sweep; it is a theorem, and +`TestWindowFactorIsAlwaysFiniteAndNonNegative` states it: a `time.Duration` is +an int64 of nanoseconds, so a positive minimum is at least 1e-9 s and any span +is at most ~9.2e9 s, making the quotient at most ~9.2e18 — large, finite, and +clamped to 1 by both spellings. The two clamps are provably interchangeable for +this one factor. Recorded here so the absence of a failing mutant reads as a +proof rather than as missing coverage. + +Also not covered: the proof pins `(*Sizer).confidence` and `weakestFactor`. It +does not pin the *call sites* — that `a.Confidence.Score` still feeds +`Proposal.Confidence` and the `low-confidence` refusal. Those are covered by +the pre-existing `TestSuppressionLowConfidenceFiresAlone` in both packages, +which assert the refusal names `metric-resolution` (ec2) and `report-coverage` +(lambda) respectively. **No existing test was modified**; 37 packages pass +`go test -race -short ./...`. + +### 4.3 The frozen copies + +The legacy implementations in the two `_test.go` files are frozen and are not +maintained alongside the real code. When a deliberate change to a confidence +model lands, this proof's job is to **fail**; the new behaviour is then +re-pinned by editing the frozen copy in the same commit that changes the model, +so the diff shows the number moving. Deleting the copies to make the test pass +deletes the only thing standing between a refactor and a silent recalibration. + +--- + +## 5. What `pkg/rds` should call + +When U13 grows a confidence model, it writes **factors and weights only**. The +machinery is: + +```go +import "github.com/agenticode/kilter/pkg/confidence" + +type ConfidenceFactor = confidence.Factor +type Confidence = confidence.Confidence + +func (s *Sizer) confidence(obs Observation) Confidence { + var c Confidence + c.Add("...", weightSomething, earned, why) // Add, never AddBounded + + // RDS quotes a multi-day CloudWatch window: round the prose to the hour. + windowEarned, windowWhy := confidence.WindowFactor( + obs.Window.Duration(), s.cfg.MinWindow, obs.Window.String(), time.Hour) + c.Add(confidence.FactorWindow, weightWindow, windowEarned, windowWhy) + return c +} +``` + +and the refusal is +`fmt.Sprintf("... %s", confidence.WeakestFactor(a.Confidence))`, with +`Proposal.Confidence = a.Confidence.Score` — still a `float64`, unchanged in +type and meaning. + +Four things `pkg/rds` must **not** do: + +1. **Do not use `AddBounded`.** It exists only to hold `pkg/ec2` still. A new + domain has no shipped numbers to preserve and should reject non-finite + evidence. +2. **Do not copy `pkg/ec2`'s or `pkg/lambda`'s weights.** They are claims about + CPU/memory metrics and Lambda REPORT lines. RDS's evidence is different — + most obviously, `FreeableMemory` is `MemAvailable` (`pkg/rds/FINDINGS.md` + §4), so a memory factor there means something else entirely and should + probably be a refusal rather than a weight. +3. **Do not reach for `decision.Compose` as well.** Pick one model. A domain + with both would produce two incomparable confidences for the same target. +4. **Do not add a `window` factor with a different name.** `FactorWindow` is a + constant because operators and tests read it out of refusals. + +Once RDS lands, the additive model has three users and one implementation — +which was the whole ask. + +--- + +## 6. Left for someone else + +- **`pkg/ec2` adopting `Add`.** §2.2. One word, one file, one test to re-pin. + Deliberately not bundled into a refactor. +- **`pkg/ecs` keeping its `Basis`.** `pkg/ecs/sizer.go:646` calls + `decision.Compose` and immediately discards everything but `.Score`, so its + `low-confidence` refusal can only say *that* confidence was low, never + *which* term cost it — the thing `WeakestFactor` gives the other two domains + for free. `decision.Confidence.Basis` is already populated. A + `decision.WeakestTerm` would be the multiplicative model's equivalent and + belongs in `pkg/decision`, not here. Out of scope this wave. +- **Reconciling the two models.** Not obviously desirable. Recorded as a real + fork with four reasons (§2.1) rather than as debt, so that anyone proposing + to merge them has to argue against those four rather than discover them. diff --git a/pkg/confidence/confidence.go b/pkg/confidence/confidence.go new file mode 100644 index 0000000..f70fa54 --- /dev/null +++ b/pkg/confidence/confidence.go @@ -0,0 +1,138 @@ +// Package confidence holds the "earned, not lost" confidence model shared by +// the cloud sizing domains. +// +// A score starts at zero and adds only what the evidence demonstrates, so a +// missing signal cannot be mistaken for a present one. Each domain still owns +// its own factors, weights and prose — those are claims about that domain's +// evidence and do not transfer. What lives here is the machinery underneath +// them: the factor record, the accumulator with its clamp, the reporting +// projection [WeakestFactor], and the one factor whose arithmetic is genuinely +// identical across domains ([WindowFactor]). +// +// This is deliberately NOT the only confidence model in the repo. +// [github.com/agenticode/kilter/pkg/decision.Compose] is a multiplicative one +// with different semantics (a zero term vetoes the score; the result is +// rounded to two decimals), and pkg/ecs and pkg/domain/fargate use it because +// they inherit the recommender's historical formula. The two models are not +// interchangeable and merging them would move shipped numbers. See +// FINDINGS.md. +package confidence + +import ( + "fmt" + "math" + "time" +) + +// Factor is one earned component of a confidence score. +// +// The JSON shape is load-bearing: it is what a report serializes and what an +// operator reads to know which measurement to go fix. +type Factor struct { + Name string `json:"name"` + Weight float64 `json:"weight"` + Earned float64 `json:"earned"` // 0..1 + Why string `json:"why"` +} + +// Confidence is a score built from nothing. It starts at zero and adds only +// what the evidence earns, so a missing signal cannot be mistaken for a +// present one; a domain's own MinConfidence is the bar it has to clear. +// +// Factors are appended in call order and never sorted, so the slice — and +// therefore [WeakestFactor] and the serialized report — is deterministic. +type Confidence struct { + Score float64 `json:"score"` + Factors []Factor `json:"factors,omitempty"` +} + +// Add appends a factor and credits weight×earned to the score. +// +// earned is clamped to [0,1], and a non-finite earned — NaN or ±Inf — is not +// evidence of anything and earns 0. That last clause is the difference from +// [Confidence.AddBounded]; prefer this one. +func (c *Confidence) Add(name string, weight, earned float64, why string) { + if !finite(earned) || earned < 0 { + earned = 0 + } + if earned > 1 { + earned = 1 + } + c.append(name, weight, earned, why) +} + +// AddBounded is [Confidence.Add] with the non-finite guard omitted: earned is +// clamped by comparison alone. NaN is neither < 0 nor > 1, so it passes +// through and poisons Score; +Inf clamps to 1 rather than earning 0. +// +// This exists because pkg/ec2 shipped with exactly this clamp and its scores +// are its shipped output. Adopting Add there is a one-word diff and a +// deliberate behaviour change, not a refactor — which is the point of keeping +// the two spellings distinguishable rather than averaging them. +func (c *Confidence) AddBounded(name string, weight, earned float64, why string) { + if earned < 0 { + earned = 0 + } + if earned > 1 { + earned = 1 + } + c.append(name, weight, earned, why) +} + +func (c *Confidence) append(name string, weight, earned float64, why string) { + c.Factors = append(c.Factors, Factor{Name: name, Weight: weight, Earned: earned, Why: why}) + c.Score += weight * earned +} + +// finite is pkg/lambda's guard, character for character, so the lift cannot +// have quietly changed which values count as evidence. +func finite(f float64) bool { return !math.IsNaN(f) && !math.IsInf(f, 0) } + +// WeakestFactor names the factor that cost the most confidence, so a +// low-confidence refusal says what would fix it rather than only that it +// failed. +// +// This is a reporting surface, not an internal detail: an operator reads it to +// know what to go measure. Loss is Weight×(1−Earned) and ties resolve to the +// earliest factor, so the answer depends only on the order factors were added. +func WeakestFactor(c Confidence) string { + worst, lost := "", -1.0 + for _, f := range c.Factors { + if l := f.Weight * (1 - f.Earned); l > lost { + worst, lost = f.Name+": "+f.Why, l + } + } + if worst == "" { + return "no single dominant factor" + } + return worst +} + +// FactorWindow is the shared name of the observation-span factor. It is a +// constant because it is printed by [WeakestFactor] into a refusal an operator +// reads and a test asserts on. +const FactorWindow = "window" + +// WindowFactor is the observation-span factor: the observed span as a fraction +// of the minimum the domain requires, and the prose saying so. +// +// The returned fraction is uncapped — clamping is [Confidence.Add]'s job, and +// routing it through there is what keeps a domain's non-finite policy in one +// place. A non-positive minimum earns nothing: a domain that cannot say how +// much window it needs has not been shown that it got enough. +// +// round controls only the prose. It is a parameter rather than a constant +// because it is a domain fact: an EC2 window is quoted against a multi-day +// minimum and rounds to the hour, a Lambda window against a minimum of hours +// and rounds to the minute. Rounding a 6-hour Lambda minimum to the hour is +// not a formatting preference, it is a wrong number. +// +// The division is deliberately Seconds()/Seconds() and not +// float64(observed)/float64(minimum): the two disagree in the last bit for +// some inputs, and this factor's value is shipped output. +func WindowFactor(observed, minimum time.Duration, observedText string, round time.Duration) (earned float64, why string) { + if minimum > 0 { + earned = observed.Seconds() / minimum.Seconds() + } + return earned, fmt.Sprintf("observed %s against a %s minimum", observedText, minimum.Round(round)) +} diff --git a/pkg/confidence/confidence_test.go b/pkg/confidence/confidence_test.go new file mode 100644 index 0000000..24ac350 --- /dev/null +++ b/pkg/confidence/confidence_test.go @@ -0,0 +1,185 @@ +package confidence + +import ( + "math" + "testing" + "time" +) + +// The domain-level proofs live in pkg/ec2 and pkg/lambda, because that is +// where the shipped numbers are. These tests pin this package's own contract: +// the two clamps, the ordering WeakestFactor depends on, and the one theorem +// that lets a domain choose either clamp for the window factor. + +func TestAddClampsIntoTheUnitInterval(t *testing.T) { + for _, tc := range []struct{ in, want float64 }{ + {0, 0}, {0.5, 0.5}, {1, 1}, + {-0.0, 0}, {-1, 0}, {-math.MaxFloat64, 0}, + {2, 1}, {math.MaxFloat64, 1}, + {math.NaN(), 0}, {math.Inf(1), 0}, {math.Inf(-1), 0}, + } { + var c Confidence + c.Add("f", 1, tc.in, "why") + if c.Factors[0].Earned != tc.want || c.Score != tc.want { + t.Errorf("Add(%v): earned %v score %v, want %v", tc.in, c.Factors[0].Earned, c.Score, tc.want) + } + } +} + +// AddBounded differs from Add on exactly two inputs. Anything else diverging +// means the two spellings have drifted apart for a reason nobody recorded. +func TestAddBoundedDiffersFromAddOnlyOnNonFiniteInput(t *testing.T) { + inputs := []float64{ + 0, 0.25, 0.5, 1, -0.0, -1, -1e300, 2, 1e300, math.MaxFloat64, + math.SmallestNonzeroFloat64, 0.9999999999999999, + math.NaN(), math.Inf(1), math.Inf(-1), + } + for _, in := range inputs { + var a, b Confidence + a.Add("f", 1, in, "why") + b.AddBounded("f", 1, in, "why") + same := math.Float64bits(a.Factors[0].Earned) == math.Float64bits(b.Factors[0].Earned) + wantDiffer := math.IsNaN(in) || math.IsInf(in, 1) + if same == wantDiffer { + t.Errorf("Add(%v)=%v AddBounded(%v)=%v: differ=%v, want differ=%v", + in, a.Factors[0].Earned, in, b.Factors[0].Earned, !same, wantDiffer) + } + } + // -Inf agrees because both clamps floor it: only NaN and +Inf diverge. + var a, b Confidence + a.Add("f", 1, math.Inf(-1), "why") + b.AddBounded("f", 1, math.Inf(-1), "why") + if a.Factors[0].Earned != 0 || b.Factors[0].Earned != 0 { + t.Errorf("-Inf: Add %v, AddBounded %v, want both 0", a.Factors[0].Earned, b.Factors[0].Earned) + } +} + +func TestWeakestFactorNamesTheLargestLossAndBreaksTiesByOrder(t *testing.T) { + var c Confidence + c.Add("first", 0.3, 0.5, "a") // loses 0.15 + c.Add("second", 0.3, 0.5, "b") // loses 0.15 — a tie + c.Add("third", 0.4, 1, "c") // loses 0 + if got, want := WeakestFactor(c), "first: a"; got != want { + t.Errorf("WeakestFactor = %q, want %q: ties resolve to the earliest factor", got, want) + } + + var big Confidence + big.Add("small", 0.9, 0.9, "x") // loses 0.09 + big.Add("large", 0.2, 0, "y") // loses 0.20 + if got, want := WeakestFactor(big), "large: y"; got != want { + t.Errorf("WeakestFactor = %q, want %q", got, want) + } +} + +// A perfect score still has to answer, because callers print the result +// unconditionally into a refusal. +func TestWeakestFactorOnPerfectAndEmptyScores(t *testing.T) { + if got := WeakestFactor(Confidence{}); got != "no single dominant factor" { + t.Errorf("empty: %q", got) + } + var c Confidence + c.Add("all", 1, 1, "everything measured") + if got, want := WeakestFactor(c), "all: everything measured"; got != want { + t.Errorf("perfect: %q, want %q — zero loss is still the weakest of one", got, want) + } +} + +// Factors are never sorted, so the slice, WeakestFactor and any serialized +// report are functions of call order alone. +func TestFactorsKeepCallOrder(t *testing.T) { + var c Confidence + names := []string{"z", "a", "m", "b"} + for _, n := range names { + c.Add(n, 0.25, 1, "why") + } + for i, n := range names { + if c.Factors[i].Name != n { + t.Fatalf("factor %d = %q, want %q", i, c.Factors[i].Name, n) + } + } +} + +// TestWindowFactorIsAlwaysFiniteAndNonNegative is the theorem that makes the +// clamp choice irrelevant for THIS factor: a domain may add it with Add or +// with AddBounded and get the same number either way. +// +// It holds because a time.Duration is an int64 of nanoseconds, so a positive +// minimum is at least 1ns (1e-9 s) and any observed span is at most ~292 +// years (~9.2e9 s). The largest possible quotient is ~9.2e18 — large, but +// finite, and clamped to 1 by both spellings. Observed spans are non-negative +// because a domain's Window.Duration() floors at zero. +// +// This is why the pkg/lambda equivalence proof does not — and cannot — +// distinguish Add from AddBounded on the window factor. Recorded so that +// absence of a failing case reads as a proof rather than as a gap. +func TestWindowFactorIsAlwaysFiniteAndNonNegative(t *testing.T) { + spans := []time.Duration{ + 0, 1, time.Second, time.Hour, 365 * 24 * time.Hour, math.MaxInt64, + } + minimums := []time.Duration{ + 1, 2, time.Nanosecond, time.Second, 7 * 24 * time.Hour, math.MaxInt64, + } + for _, s := range spans { + for _, m := range minimums { + earned, _ := WindowFactor(s, m, "text", time.Hour) + if math.IsNaN(earned) || math.IsInf(earned, 0) || earned < 0 { + t.Fatalf("WindowFactor(%v, %v) = %v, want finite and non-negative", s, m, earned) + } + var a, b Confidence + a.Add(FactorWindow, 0.2, earned, "w") + b.AddBounded(FactorWindow, 0.2, earned, "w") + if math.Float64bits(a.Score) != math.Float64bits(b.Score) { + t.Fatalf("WindowFactor(%v, %v): Add %v != AddBounded %v", s, m, a.Score, b.Score) + } + } + } +} + +// A non-positive minimum earns nothing but still prints, because the prose is +// what tells an operator the config is the problem. +func TestWindowFactorWithNoStatedMinimumEarnsNothing(t *testing.T) { + for _, m := range []time.Duration{0, -time.Hour} { + earned, why := WindowFactor(24*time.Hour, m, "24h0m0s", time.Hour) + if earned != 0 { + t.Errorf("minimum %v earned %v, want 0", m, earned) + } + if why == "" { + t.Errorf("minimum %v produced no prose", m) + } + } +} + +// The rounding argument is a domain fact, not a formatting preference: the +// same span reads differently for a domain whose minimum is days and one +// whose minimum is hours. +func TestWindowFactorRoundingIsCallerChosen(t *testing.T) { + hourly, _ := WindowFactor(time.Hour, 6*time.Hour+20*time.Minute, "1h0m0s", time.Hour) + minutely, _ := WindowFactor(time.Hour, 6*time.Hour+20*time.Minute, "1h0m0s", time.Minute) + if math.Float64bits(hourly) != math.Float64bits(minutely) { + t.Fatal("rounding must not touch the earned value") + } + _, hourWhy := WindowFactor(time.Hour, 6*time.Hour+20*time.Minute, "1h0m0s", time.Hour) + _, minWhy := WindowFactor(time.Hour, 6*time.Hour+20*time.Minute, "1h0m0s", time.Minute) + if hourWhy == minWhy { + t.Fatalf("rounding did not change the prose: both %q", hourWhy) + } + if want := "observed 1h0m0s against a 6h20m0s minimum"; minWhy != want { + t.Errorf("minute rounding: %q, want %q", minWhy, want) + } + if want := "observed 1h0m0s against a 6h0m0s minimum"; hourWhy != want { + t.Errorf("hour rounding: %q, want %q", hourWhy, want) + } +} + +// Score is a plain sum in call order, so it is reproducible; nothing here +// iterates a map or reads a clock. +func TestScoreIsTheWeightedSumInCallOrder(t *testing.T) { + var c Confidence + c.Add("a", 0.3, 1, "") + c.Add("b", 0.2, 0.5, "") + c.Add("c", 0.5, 0, "") + want := 0.3*1 + 0.2*0.5 + 0.5*0 + if math.Float64bits(c.Score) != math.Float64bits(want) { + t.Errorf("score %v, want %v", c.Score, want) + } +} diff --git a/pkg/ec2/confidence_equiv_test.go b/pkg/ec2/confidence_equiv_test.go new file mode 100644 index 0000000..a376c2f --- /dev/null +++ b/pkg/ec2/confidence_equiv_test.go @@ -0,0 +1,222 @@ +package ec2 + +import ( + "fmt" + "math" + "testing" + "time" +) + +// The confidence model moved to pkg/confidence. This file is the proof that +// the move changed nothing: it carries the pre-lift implementation verbatim +// and asserts that the shipped path reproduces it BIT for bit — Score, every +// factor field, and the weakestFactor prose an operator reads out of a +// low-confidence refusal. +// +// The copies below are frozen. They are not maintained alongside the real +// implementation and must not be "fixed"; the moment a deliberate change to +// the model lands, this file's job is to fail, and the new behaviour is then +// re-pinned by editing the copies in the same commit that changes the model. + +type legacyFactor struct { + Name string + Weight float64 + Earned float64 + Why string +} + +type legacyConfidence struct { + Score float64 + Factors []legacyFactor +} + +// legacyAdd is pkg/ec2's pre-lift add, character for character. Note what it +// does NOT do: NaN is neither < 0 nor > 1, so it survives both comparisons and +// poisons Score. That is the shipped behaviour, and the table below pins it. +func (c *legacyConfidence) add(name string, weight, earned float64, why string) { + if earned < 0 { + earned = 0 + } + if earned > 1 { + earned = 1 + } + c.Factors = append(c.Factors, legacyFactor{Name: name, Weight: weight, Earned: earned, Why: why}) + c.Score += weight * earned +} + +// legacyConfidenceOf is the pre-lift (*Sizer).confidence, verbatim. +func legacyConfidenceOf(cfg Config, obs Observation) legacyConfidence { + var c legacyConfidence + c.add("sample-coverage", weightCoverage, obs.Coverage, + fmt.Sprintf("%d of ~%d expected datapoints", obs.Samples, obs.ExpectedSamples)) + + windowEarned := 0.0 + if cfg.MinWindow > 0 { + windowEarned = obs.Window.Duration().Seconds() / cfg.MinWindow.Seconds() + } + c.add("window", weightWindow, windowEarned, + fmt.Sprintf("observed %s against a %s minimum", obs.Window.String(), cfg.MinWindow.Round(time.Hour))) + + memEarned, memWhy := 1.0, "memory observed via the CloudWatch agent" + if obs.MemoryBlind { + memEarned, memWhy = 0, "memory-blind: no CloudWatch agent, so no memory metric exists" + } + c.add("memory-signal", weightMemory, memEarned, memWhy) + + resEarned, resWhy := 1.0, fmt.Sprintf("%d-second datapoints", obs.PeriodSeconds) + if obs.PeriodSeconds > PeriodDetailedSeconds { + resEarned = 0.4 + resWhy = fmt.Sprintf("%d-second datapoints hide shorter peaks", obs.PeriodSeconds) + } + c.add("metric-resolution", weightResolution, resEarned, resWhy) + + burstEarned, burstWhy := 1.0, "not a credit-based instance type" + switch obs.Burst.Class { + case BurstUnknown: + burstEarned, burstWhy = 0, "burstable with no usable credit evidence" + case BurstThrottled: + burstEarned, burstWhy = 0, "credit-depleted: observed CPU is a throttling ceiling" + case BurstHealthy, BurstSurplus: + burstWhy = "credit metrics present and classified" + } + c.add("burst-evidence", weightBurst, burstEarned, burstWhy) + return c +} + +// legacyWeakestFactor is pkg/ec2's pre-lift weakestFactor, verbatim. +func legacyWeakestFactor(c legacyConfidence) string { + worst, lost := "", -1.0 + for _, f := range c.Factors { + if l := f.Weight * (1 - f.Earned); l > lost { + worst, lost = f.Name+": "+f.Why, l + } + } + if worst == "" { + return "no single dominant factor" + } + return worst +} + +// win builds a window of exactly d from a fixed instant. No clock is read: +// confidence is pure and this test must be reproducible forever. +func win(d time.Duration) Window { + start := time.Date(2024, 3, 1, 0, 0, 0, 0, time.UTC) + return Window{Start: start, End: start.Add(d)} +} + +// TestConfidenceEquivalenceAfterLift is the acceptance criterion for moving +// this model into pkg/confidence: every input produces the identical score, +// the identical factor list and the identical weakest-factor prose. +func TestConfidenceEquivalenceAfterLift(t *testing.T) { + windows := []time.Duration{ + 0, time.Minute, 90 * time.Minute, 24 * time.Hour, + 7 * 24 * time.Hour, 14 * 24 * time.Hour, 400 * 24 * time.Hour, + // A span whose Seconds() is not exactly representable, so a lift + // that swapped Seconds()/Seconds() for float64/float64 would show up. + 3*time.Hour + 7*time.Minute + 13*time.Second + 456789123*time.Nanosecond, + } + minWindows := []time.Duration{ + 0, -time.Hour, time.Nanosecond, 45 * time.Minute, + 7 * 24 * time.Hour, 3*time.Hour + 20*time.Minute, + } + coverages := []float64{ + 0, 0.5, 1, 0.9999999999999999, 1e-300, + // Out of band and non-finite: the clamp is the one place ec2 and + // lambda genuinely disagree, so it is pinned hardest. + -1, -0.0, 2, math.MaxFloat64, + math.NaN(), math.Inf(1), math.Inf(-1), + } + periods := []int32{0, -60, 60, PeriodDetailedSeconds, PeriodDetailedSeconds + 1, 300, 3600} + classes := []BurstClass{ + BurstNotApplicable, BurstUnknown, BurstHealthy, BurstThrottled, BurstSurplus, + "", "some-future-class", + } + + cases := 0 + for _, mw := range minWindows { + for _, w := range windows { + for _, cov := range coverages { + for _, p := range periods { + for _, cl := range classes { + for _, blind := range []bool{false, true} { + cfg := Config{MinWindow: mw} + obs := Observation{ + Window: win(w), + PeriodSeconds: p, + Samples: int(cov * 1000), + ExpectedSamples: 1000, + Coverage: cov, + MemoryBlind: blind, + Burst: BurstState{Class: cl}, + } + s := &Sizer{cfg: cfg} + cases++ + assertSameConfidence(t, legacyConfidenceOf(cfg, obs), s.confidence(obs)) + if t.Failed() { + t.Fatalf("first divergence at MinWindow=%v Window=%v Coverage=%v Period=%d Burst=%q Blind=%v", + mw, w, cov, p, cl, blind) + } + } + } + } + } + } + } + if cases < 10000 { + t.Fatalf("only %d cases exercised; the proof is meant to be dense", cases) + } + t.Logf("%d input combinations produce bit-identical confidence", cases) +} + +// assertSameConfidence compares by raw bits, not by tolerance. A confidence +// that moves in the last bit is still a confidence that moved. +func assertSameConfidence(t *testing.T, want legacyConfidence, got Confidence) { + t.Helper() + if math.Float64bits(want.Score) != math.Float64bits(got.Score) { + t.Errorf("score: legacy %v (%#x) != lifted %v (%#x)", + want.Score, math.Float64bits(want.Score), got.Score, math.Float64bits(got.Score)) + } + if len(want.Factors) != len(got.Factors) { + t.Fatalf("factor count: legacy %d != lifted %d", len(want.Factors), len(got.Factors)) + } + for i, wf := range want.Factors { + gf := got.Factors[i] + if wf.Name != gf.Name { + t.Errorf("factor %d name: %q != %q", i, wf.Name, gf.Name) + } + if math.Float64bits(wf.Weight) != math.Float64bits(gf.Weight) { + t.Errorf("factor %d (%s) weight: %v != %v", i, wf.Name, wf.Weight, gf.Weight) + } + if math.Float64bits(wf.Earned) != math.Float64bits(gf.Earned) { + t.Errorf("factor %d (%s) earned: %v (%#x) != %v (%#x)", i, wf.Name, + wf.Earned, math.Float64bits(wf.Earned), gf.Earned, math.Float64bits(gf.Earned)) + } + if wf.Why != gf.Why { + t.Errorf("factor %d (%s) why:\n legacy %q\n lifted %q", i, wf.Name, wf.Why, gf.Why) + } + } + // weakestFactor is a reporting surface: an operator reads it to know what + // to go measure, so a change in WHICH factor is named is a behaviour + // change even when the number is unchanged. + if w, g := legacyWeakestFactor(want), weakestFactor(got); w != g { + t.Errorf("weakestFactor:\n legacy %q\n lifted %q", w, g) + } +} + +// TestConfidenceKeepsItsNaNPropagation pins the divergence between this +// domain's clamp and pkg/lambda's, so that closing it can only ever be a +// deliberate, visible change rather than a side effect of a refactor. +func TestConfidenceKeepsItsNaNPropagation(t *testing.T) { + s := &Sizer{cfg: Config{MinWindow: 7 * 24 * time.Hour}} + c := s.confidence(Observation{Window: win(7 * 24 * time.Hour), Coverage: math.NaN()}) + if !math.IsNaN(c.Factors[0].Earned) { + t.Errorf("sample-coverage earned = %v, want NaN: pkg/ec2 clamps by comparison alone", c.Factors[0].Earned) + } + if !math.IsNaN(c.Score) { + t.Errorf("score = %v, want NaN", c.Score) + } + inf := s.confidence(Observation{Window: win(7 * 24 * time.Hour), Coverage: math.Inf(1)}) + if inf.Factors[0].Earned != 1 { + t.Errorf("+Inf coverage earned = %v, want 1 (clamped up, not zeroed)", inf.Factors[0].Earned) + } +} diff --git a/pkg/ec2/sizer.go b/pkg/ec2/sizer.go index 4de1c91..33370dc 100644 --- a/pkg/ec2/sizer.go +++ b/pkg/ec2/sizer.go @@ -6,6 +6,7 @@ import ( "sort" "time" + "github.com/agenticode/kilter/pkg/confidence" "github.com/agenticode/kilter/pkg/model" "github.com/agenticode/kilter/pkg/pricing" "github.com/agenticode/kilter/pkg/pricing/commit" @@ -133,31 +134,18 @@ type Observation struct { } // ConfidenceFactor is one earned component of a confidence score. -type ConfidenceFactor struct { - Name string `json:"name"` - Weight float64 `json:"weight"` - Earned float64 `json:"earned"` // 0..1 - Why string `json:"why"` -} +// +// Aliased rather than redeclared: the shape and its JSON are shared with +// pkg/lambda, and an alias means a report cannot drift between them. +type ConfidenceFactor = confidence.Factor // Confidence is a score built from nothing. It starts at zero and adds only // what the evidence earns, so a missing signal cannot be mistaken for a // present one; [Config.MinConfidence] is the bar it has to clear. -type Confidence struct { - Score float64 `json:"score"` - Factors []ConfidenceFactor `json:"factors,omitempty"` -} - -func (c *Confidence) add(name string, weight, earned float64, why string) { - if earned < 0 { - earned = 0 - } - if earned > 1 { - earned = 1 - } - c.Factors = append(c.Factors, ConfidenceFactor{Name: name, Weight: weight, Earned: earned, Why: why}) - c.Score += weight * earned -} +// +// This domain adds its factors with [confidence.Confidence.AddBounded], not +// Add: see pkg/confidence/FINDINGS.md §3. +type Confidence = confidence.Confidence // Confidence weights. They sum to 1. const ( @@ -857,28 +845,27 @@ func (s *Sizer) gravitonAdvisory(in Instance, cur pricing.InstanceType, obs Obse // what it can demonstrate. func (s *Sizer) confidence(obs Observation) Confidence { var c Confidence - c.add("sample-coverage", weightCoverage, obs.Coverage, + c.AddBounded("sample-coverage", weightCoverage, obs.Coverage, fmt.Sprintf("%d of ~%d expected datapoints", obs.Samples, obs.ExpectedSamples)) - windowEarned := 0.0 - if s.cfg.MinWindow > 0 { - windowEarned = obs.Window.Duration().Seconds() / s.cfg.MinWindow.Seconds() - } - c.add("window", weightWindow, windowEarned, - fmt.Sprintf("observed %s against a %s minimum", obs.Window.String(), s.cfg.MinWindow.Round(time.Hour))) + // An EC2 minimum window is quoted in days, so the prose rounds to the + // hour. That argument is the domain fact pkg/confidence refuses to guess. + windowEarned, windowWhy := confidence.WindowFactor( + obs.Window.Duration(), s.cfg.MinWindow, obs.Window.String(), time.Hour) + c.AddBounded(confidence.FactorWindow, weightWindow, windowEarned, windowWhy) memEarned, memWhy := 1.0, "memory observed via the CloudWatch agent" if obs.MemoryBlind { memEarned, memWhy = 0, "memory-blind: no CloudWatch agent, so no memory metric exists" } - c.add("memory-signal", weightMemory, memEarned, memWhy) + c.AddBounded("memory-signal", weightMemory, memEarned, memWhy) resEarned, resWhy := 1.0, fmt.Sprintf("%d-second datapoints", obs.PeriodSeconds) if obs.PeriodSeconds > PeriodDetailedSeconds { resEarned = 0.4 resWhy = fmt.Sprintf("%d-second datapoints hide shorter peaks", obs.PeriodSeconds) } - c.add("metric-resolution", weightResolution, resEarned, resWhy) + c.AddBounded("metric-resolution", weightResolution, resEarned, resWhy) burstEarned, burstWhy := 1.0, "not a credit-based instance type" switch obs.Burst.Class { @@ -889,24 +876,13 @@ func (s *Sizer) confidence(obs Observation) Confidence { case BurstHealthy, BurstSurplus: burstWhy = "credit metrics present and classified" } - c.add("burst-evidence", weightBurst, burstEarned, burstWhy) + c.AddBounded("burst-evidence", weightBurst, burstEarned, burstWhy) return c } // weakestFactor names the factor that cost the most confidence, so a // low-confidence refusal says what would fix it. -func weakestFactor(c Confidence) string { - worst, lost := "", -1.0 - for _, f := range c.Factors { - if l := f.Weight * (1 - f.Earned); l > lost { - worst, lost = f.Name+": "+f.Why, l - } - } - if worst == "" { - return "no single dominant factor" - } - return worst -} +func weakestFactor(c Confidence) string { return confidence.WeakestFactor(c) } func seriesStatus(t Target) string { for _, s := range t.Series { diff --git a/pkg/lambda/confidence_equiv_test.go b/pkg/lambda/confidence_equiv_test.go new file mode 100644 index 0000000..49d4089 --- /dev/null +++ b/pkg/lambda/confidence_equiv_test.go @@ -0,0 +1,236 @@ +package lambda + +import ( + "fmt" + "math" + "testing" + "time" +) + +// The confidence model moved to pkg/confidence. This file is the proof that +// the move changed nothing: it carries the pre-lift implementation verbatim +// and asserts that the shipped path reproduces it BIT for bit — Score, every +// factor field, and the weakestFactor prose an operator reads out of a +// low-confidence refusal. +// +// The copies below are frozen; see the same file in pkg/ec2 for why. + +type legacyFactor struct { + Name string + Weight float64 + Earned float64 + Why string +} + +type legacyConfidence struct { + Score float64 + Factors []legacyFactor +} + +// legacyAdd is pkg/lambda's pre-lift add, character for character. Unlike +// pkg/ec2's it zeroes a non-finite earned value rather than letting NaN +// through — the one factor-level disagreement between the two domains. +func (c *legacyConfidence) add(name string, weight, earned float64, why string) { + if !finite(earned) || earned < 0 { + earned = 0 + } + if earned > 1 { + earned = 1 + } + c.Factors = append(c.Factors, legacyFactor{Name: name, Weight: weight, Earned: earned, Why: why}) + c.Score += weight * earned +} + +// legacyConfidenceOf is the pre-lift (*Sizer).confidence, verbatim. +func legacyConfidenceOf(cfg Config, obs Observation, usable []MemoryPoint) legacyConfidence { + var c legacyConfidence + + pointsEarned, pointsWhy := 0.0, fmt.Sprintf("%d memory setting(s) measured with >= %d warm invocations", + len(usable), cfg.MinSamplesPerPoint) + switch { + case len(usable) >= 3: + pointsEarned = 1 + case len(usable) == 2: + pointsEarned = 0.8 + } + c.add("measured-points", weightMeasuredPoints, pointsEarned, pointsWhy) + + c.add("report-coverage", weightReportCoverage, obs.ReportCoverage, + fmt.Sprintf("%d REPORT lines parsed for %.0f invocations (source: %s)", + obs.Records, obs.Invocations, obs.InvocationSource)) + + warmEarned, warmWhy := 0.0, "no invocations observed" + if obs.Warm+obs.Cold > 0 { + warmEarned = 1 - obs.ColdShare + warmWhy = fmt.Sprintf("%.1f%% of invocations were cold starts", obs.ColdShare*100) + } + c.add("warm-share", weightWarmShare, warmEarned, warmWhy) + + windowEarned := 0.0 + if cfg.MinWindow > 0 { + windowEarned = obs.Window.Duration().Seconds() / cfg.MinWindow.Seconds() + } + c.add("window", weightWindow, windowEarned, + fmt.Sprintf("observed %s against a %s minimum", obs.Window.String(), cfg.MinWindow.Round(time.Minute))) + + headEarned, headWhy := 0.0, "max memory used is at the configured ceiling: possibly truncated" + if cur, ok := obs.Current(); ok && cur.MemoryMB > 0 && !cur.AtCeiling { + margin := 1 - float64(cur.MaxMemoryUsedMB)/float64(cur.MemoryMB) + headEarned = math.Min(1, margin/0.25) + headWhy = fmt.Sprintf("%s used of %s configured (%.0f%% margin)", + fmtMB(cur.MaxMemoryUsedMB), fmtMB(cur.MemoryMB), margin*100) + } + c.add("memory-headroom", weightHeadroom, headEarned, headWhy) + return c +} + +// legacyWeakestFactor is pkg/lambda's pre-lift weakestFactor, verbatim. +func legacyWeakestFactor(c legacyConfidence) string { + worst, lost := "", -1.0 + for _, f := range c.Factors { + if l := f.Weight * (1 - f.Earned); l > lost { + worst, lost = f.Name+": "+f.Why, l + } + } + if worst == "" { + return "no single dominant factor" + } + return worst +} + +// equivWindow builds a window of exactly d from a fixed instant. No clock is +// read: confidence is pure and this test must be reproducible forever. +func equivWindow(d time.Duration) Window { + start := time.Date(2024, 3, 1, 0, 0, 0, 0, time.UTC) + return Window{Start: start, End: start.Add(d)} +} + +// TestConfidenceEquivalenceAfterLift is the acceptance criterion for moving +// this model into pkg/confidence: every input produces the identical score, +// the identical factor list and the identical weakest-factor prose. +func TestConfidenceEquivalenceAfterLift(t *testing.T) { + windows := []time.Duration{ + 0, time.Second, 30 * time.Minute, 6 * time.Hour, 24 * time.Hour, 30 * 24 * time.Hour, + // A span whose Seconds() is not exactly representable, so a lift that + // swapped Seconds()/Seconds() for float64/float64 would show up. + 3*time.Hour + 7*time.Minute + 13*time.Second + 456789123*time.Nanosecond, + } + minWindows := []time.Duration{0, -time.Hour, time.Nanosecond, 6 * time.Hour, 3*time.Hour + 20*time.Minute} + coverages := []float64{ + 0, 0.25, 1, 0.9999999999999999, + // Out of band and non-finite: this domain zeroes what pkg/ec2 keeps. + -1, 2, math.NaN(), math.Inf(1), math.Inf(-1), + } + coldShares := []float64{0, 0.5, 1, 1.5, -0.5, math.NaN(), math.Inf(1)} + // Points exercise every branch of measured-points (0/1/2/3+) and of + // memory-headroom (no current point, zero memory, at ceiling, margins + // above and below the 25% saturation, and a negative margin). + pointSets := [][]MemoryPoint{ + nil, + {{MemoryMB: 512, MaxMemoryUsedMB: 100, Warm: 10}}, + {{MemoryMB: 512, MaxMemoryUsedMB: 500, Warm: 10}, {MemoryMB: 1024, MaxMemoryUsedMB: 500, Warm: 10}}, + {{MemoryMB: 512}, {MemoryMB: 1024}, {MemoryMB: 2048}}, + {{MemoryMB: 0, MaxMemoryUsedMB: 0}}, + {{MemoryMB: 512, MaxMemoryUsedMB: 512, AtCeiling: true}}, + {{MemoryMB: 512, MaxMemoryUsedMB: 600}}, + {{MemoryMB: 1024, MaxMemoryUsedMB: 1000}}, + {{MemoryMB: 1769, MaxMemoryUsedMB: 3}}, + } + + cases := 0 + for _, mw := range minWindows { + for _, w := range windows { + for _, cov := range coverages { + for _, cs := range coldShares { + for pi, pts := range pointSets { + for _, idx := range []int{-1, 0, len(pts) - 1, len(pts)} { + for _, warm := range []int{0, 700} { + cfg := Config{MinWindow: mw, MinSamplesPerPoint: 50} + obs := Observation{ + Window: equivWindow(w), + Records: 1400, + Warm: warm, + Cold: 0, + ColdShare: cs, + Invocations: 2_000_000, + InvocationSource: SourceCloudWatch, + ReportCoverage: cov, + Points: pts, + CurrentIndex: idx, + } + // usable is passed in by the caller, so vary it + // independently of Points to cover every branch. + usable := pts + if pi%2 == 1 { + usable = nil + } + s := &Sizer{cfg: cfg} + cases++ + assertSameConfidence(t, + legacyConfidenceOf(cfg, obs, usable), s.confidence(obs, usable)) + if t.Failed() { + t.Fatalf("first divergence at MinWindow=%v Window=%v Coverage=%v ColdShare=%v points=%d idx=%d warm=%d", + mw, w, cov, cs, pi, idx, warm) + } + } + } + } + } + } + } + } + if cases < 10000 { + t.Fatalf("only %d cases exercised; the proof is meant to be dense", cases) + } + t.Logf("%d input combinations produce bit-identical confidence", cases) +} + +// assertSameConfidence compares by raw bits, not by tolerance. A confidence +// that moves in the last bit is still a confidence that moved. +func assertSameConfidence(t *testing.T, want legacyConfidence, got Confidence) { + t.Helper() + if math.Float64bits(want.Score) != math.Float64bits(got.Score) { + t.Errorf("score: legacy %v (%#x) != lifted %v (%#x)", + want.Score, math.Float64bits(want.Score), got.Score, math.Float64bits(got.Score)) + } + if len(want.Factors) != len(got.Factors) { + t.Fatalf("factor count: legacy %d != lifted %d", len(want.Factors), len(got.Factors)) + } + for i, wf := range want.Factors { + gf := got.Factors[i] + if wf.Name != gf.Name { + t.Errorf("factor %d name: %q != %q", i, wf.Name, gf.Name) + } + if math.Float64bits(wf.Weight) != math.Float64bits(gf.Weight) { + t.Errorf("factor %d (%s) weight: %v != %v", i, wf.Name, wf.Weight, gf.Weight) + } + if math.Float64bits(wf.Earned) != math.Float64bits(gf.Earned) { + t.Errorf("factor %d (%s) earned: %v (%#x) != %v (%#x)", i, wf.Name, + wf.Earned, math.Float64bits(wf.Earned), gf.Earned, math.Float64bits(gf.Earned)) + } + if wf.Why != gf.Why { + t.Errorf("factor %d (%s) why:\n legacy %q\n lifted %q", i, wf.Name, wf.Why, gf.Why) + } + } + // weakestFactor is a reporting surface: an operator reads it to know what + // to go measure, so a change in WHICH factor is named is a behaviour + // change even when the number is unchanged. + if w, g := legacyWeakestFactor(want), weakestFactor(got); w != g { + t.Errorf("weakestFactor:\n legacy %q\n lifted %q", w, g) + } +} + +// TestConfidenceZeroesNonFiniteEvidence pins the other half of the divergence +// pkg/ec2 preserves: here a NaN or infinite earned value is no evidence. +func TestConfidenceZeroesNonFiniteEvidence(t *testing.T) { + s := &Sizer{cfg: Config{MinWindow: 6 * time.Hour, MinSamplesPerPoint: 50}} + for _, v := range []float64{math.NaN(), math.Inf(1), math.Inf(-1)} { + c := s.confidence(Observation{Window: equivWindow(6 * time.Hour), ReportCoverage: v, CurrentIndex: -1}, nil) + if c.Factors[1].Earned != 0 { + t.Errorf("report-coverage %v earned %v, want 0", v, c.Factors[1].Earned) + } + if math.IsNaN(c.Score) { + t.Errorf("report-coverage %v poisoned the score", v) + } + } +} diff --git a/pkg/lambda/sizer.go b/pkg/lambda/sizer.go index e685c96..edbb4db 100644 --- a/pkg/lambda/sizer.go +++ b/pkg/lambda/sizer.go @@ -6,6 +6,7 @@ import ( "sort" "time" + "github.com/agenticode/kilter/pkg/confidence" "github.com/agenticode/kilter/pkg/domain" "github.com/agenticode/kilter/pkg/pricing/commit" ) @@ -87,31 +88,18 @@ func (c Config) validate() error { } // ConfidenceFactor is one earned component of a confidence score. -type ConfidenceFactor struct { - Name string `json:"name"` - Weight float64 `json:"weight"` - Earned float64 `json:"earned"` // 0..1 - Why string `json:"why"` -} +// +// Aliased rather than redeclared: the shape and its JSON are shared with +// pkg/ec2, and an alias means a report cannot drift between them. +type ConfidenceFactor = confidence.Factor // Confidence is a score built from nothing. It starts at zero and adds only // what the evidence earns, so a missing signal cannot be mistaken for a present // one; [Config.MinConfidence] is the bar it has to clear. -type Confidence struct { - Score float64 `json:"score"` - Factors []ConfidenceFactor `json:"factors,omitempty"` -} - -func (c *Confidence) add(name string, weight, earned float64, why string) { - if !finite(earned) || earned < 0 { - earned = 0 - } - if earned > 1 { - earned = 1 - } - c.Factors = append(c.Factors, ConfidenceFactor{Name: name, Weight: weight, Earned: earned, Why: why}) - c.Score += weight * earned -} +// +// This domain adds its factors with [confidence.Confidence.Add], which treats +// a non-finite earned value as no evidence: see pkg/confidence/FINDINGS.md §3. +type Confidence = confidence.Confidence // Confidence weights. They sum to 1. measured-points carries the largest // weight on purpose: it is the factor this whole package is about. @@ -745,9 +733,9 @@ func (s *Sizer) confidence(obs Observation, usable []MemoryPoint) Confidence { case len(usable) == 2: pointsEarned = 0.8 } - c.add("measured-points", weightMeasuredPoints, pointsEarned, pointsWhy) + c.Add("measured-points", weightMeasuredPoints, pointsEarned, pointsWhy) - c.add("report-coverage", weightReportCoverage, obs.ReportCoverage, + c.Add("report-coverage", weightReportCoverage, obs.ReportCoverage, fmt.Sprintf("%d REPORT lines parsed for %.0f invocations (source: %s)", obs.Records, obs.Invocations, obs.InvocationSource)) @@ -759,14 +747,14 @@ func (s *Sizer) confidence(obs Observation, usable []MemoryPoint) Confidence { warmEarned = 1 - obs.ColdShare warmWhy = fmt.Sprintf("%.1f%% of invocations were cold starts", obs.ColdShare*100) } - c.add("warm-share", weightWarmShare, warmEarned, warmWhy) + c.Add("warm-share", weightWarmShare, warmEarned, warmWhy) - windowEarned := 0.0 - if s.cfg.MinWindow > 0 { - windowEarned = obs.Window.Duration().Seconds() / s.cfg.MinWindow.Seconds() - } - c.add("window", weightWindow, windowEarned, - fmt.Sprintf("observed %s against a %s minimum", obs.Window.String(), s.cfg.MinWindow.Round(time.Minute))) + // A Lambda log window is a slice of hours, not days, so the prose rounds + // to the minute. That argument is the domain fact pkg/confidence refuses + // to guess: rounding this minimum to the hour would print a wrong number. + windowEarned, windowWhy := confidence.WindowFactor( + obs.Window.Duration(), s.cfg.MinWindow, obs.Window.String(), time.Minute) + c.Add(confidence.FactorWindow, weightWindow, windowEarned, windowWhy) headEarned, headWhy := 0.0, "max memory used is at the configured ceiling: possibly truncated" if cur, ok := obs.Current(); ok && cur.MemoryMB > 0 && !cur.AtCeiling { @@ -775,24 +763,13 @@ func (s *Sizer) confidence(obs Observation, usable []MemoryPoint) Confidence { headWhy = fmt.Sprintf("%s used of %s configured (%.0f%% margin)", fmtMB(cur.MaxMemoryUsedMB), fmtMB(cur.MemoryMB), margin*100) } - c.add("memory-headroom", weightHeadroom, headEarned, headWhy) + c.Add("memory-headroom", weightHeadroom, headEarned, headWhy) return c } // weakestFactor names the factor that cost the most confidence, so a // low-confidence refusal says what would fix it. -func weakestFactor(c Confidence) string { - worst, lost := "", -1.0 - for _, f := range c.Factors { - if l := f.Weight * (1 - f.Earned); l > lost { - worst, lost = f.Name+": "+f.Why, l - } - } - if worst == "" { - return "no single dominant factor" - } - return worst -} +func weakestFactor(c Confidence) string { return confidence.WeakestFactor(c) } // memoryPointsValue renders the measured settings as evidence prose. func memoryPointsValue(points []MemoryPoint) string {