diff --git a/.github/workflows/release-please.yaml b/.github/workflows/release-please.yaml new file mode 100644 index 00000000..4b95c325 --- /dev/null +++ b/.github/workflows/release-please.yaml @@ -0,0 +1,35 @@ +name: Run Release Please + +# Releases everything versioned in this repository. Two packages are configured in +# release-please-config.json, each with its own release pull request: +# +# . the specification, tagged vX.Y.Z as it always has been +# specification/assets/provider-tck the provider conformance assets, tagged with their path +# +# The two are independent. The specification package excludes the conformance assets, so a +# change confined to the suite never bumps the specification; the assets carry their own +# version because four language suites pin a revision of them, and a pinned revision needs a +# name a human can read and a bot can compare. +# +# This repository enforces DCO, so the bot's own commits need a sign-off. That is the signoff +# key in release-please-config.json: v5 takes it from the configuration file, and an action +# input of the same name is not declared and would be silently ignored. + +on: + push: + branches: + - main + +permissions: + contents: write + issues: write + pull-requests: write + +jobs: + release-please: + runs-on: ubuntu-latest + steps: + - uses: googleapis/release-please-action@v5 + with: + token: ${{ secrets.GITHUB_TOKEN }} + target-branch: main diff --git a/.release-please-manifest.json b/.release-please-manifest.json new file mode 100644 index 00000000..8a76893c --- /dev/null +++ b/.release-please-manifest.json @@ -0,0 +1,4 @@ +{ + ".": "0.9.0", + "specification/assets/provider-tck": "0.0.0" +} diff --git a/release-please-config.json b/release-please-config.json new file mode 100644 index 00000000..b346fefd --- /dev/null +++ b/release-please-config.json @@ -0,0 +1,28 @@ +{ + "$schema": "https://raw.githubusercontent.com/googleapis/release-please/main/schemas/config.json", + "signoff": "OpenFeature Bot <109696520+openfeaturebot@users.noreply.github.com>", + "tag-separator": "/", + "bootstrap-sha": "dd235837f7b2ddb1120ab6e1bde940bf4d532313", + "separate-pull-requests": true, + "packages": { + ".": { + "release-type": "simple", + "package-name": "spec", + "include-component-in-tag": false, + "changelog-path": "CHANGELOG.md", + "bump-minor-pre-major": true, + "versioning": "default", + "exclude-paths": ["specification/assets/provider-tck"], + "extra-files": [] + }, + "specification/assets/provider-tck": { + "release-type": "go", + "package-name": "specification/assets/provider-tck", + "include-component-in-tag": true, + "changelog-path": "CHANGELOG.md", + "bump-minor-pre-major": true, + "versioning": "default", + "extra-files": [] + } + } +} diff --git a/specification/assets/provider-tck/.gitattributes b/specification/assets/provider-tck/.gitattributes new file mode 100644 index 00000000..e1bb310b --- /dev/null +++ b/specification/assets/provider-tck/.gitattributes @@ -0,0 +1,10 @@ +# These artifacts are consumed byte for byte by every language's provider TCK, and several +# are copied verbatim into published build artifacts. Normalise to LF so a checkout on +# Windows does not produce a different packaged file than one on Linux. +*.feature text eol=lf +*.json text eol=lf +*.yaml text eol=lf + +# embed.go ships inside the Go module zip, whose checksum is content-addressed, so it too +# must be identical whichever platform it is committed from. +*.go text eol=lf diff --git a/specification/assets/provider-tck/README.md b/specification/assets/provider-tck/README.md new file mode 100644 index 00000000..98d0d57f --- /dev/null +++ b/specification/assets/provider-tck/README.md @@ -0,0 +1,124 @@ +# Provider Conformance Assets + +Test assets for the provider conformance suite tracked in +[open-feature/spec#417](https://github.com/open-feature/spec/issues/417). + +**These assets ship ahead of their normative description.** Appendix F, which states what a TCK +implementation must do and what each capability asserts, is a separate pull request. Until it +lands this file is the reference for the capability vocabulary, and everything here is +**experimental**: the scenario set is a representative subset and the vocabulary may still change. + +These validate a **provider** against a real backend. For assets that validate an **SDK**, see [`../gherkin/`](../gherkin/README.md) and [Appendix B](../../appendix-b-gherkin-suites.md). + +## Contents + +| Path | What it is | +| --- | --- | +| [`gherkin/evaluation.feature`](./gherkin/evaluation.feature) | resolving each type with the right value and reason; the variant where the backend names one; falsy values; integer precision | +| [`gherkin/errors.feature`](./gherkin/errors.feature) | the type-mismatch matrix, numeric coercion, string typing and the unknown-flag case | +| [`gherkin/events.feature`](./gherkin/events.feature) | configuration change, and the stale/ready transition across an outage | +| [`gherkin/lifecycle.feature`](./gherkin/lifecycle.feature) | initialisation against a healthy backend and against an unreachable one; shutdown | +| [`gherkin/metadata.feature`](./gherkin/metadata.feature) | the provider identifies itself by name | +| [`gherkin/reason.feature`](./gherkin/reason.feature) | the standard resolution reasons, gated behind `@standard-reasons` | +| [`flags/canonical-flags.json`](./flags/canonical-flags.json) | the flag set every scenario assumes | +| [`openapi/control-api.yaml`](./openapi/control-api.yaml) | the HTTP surface a backend under test must expose | + +## Capabilities: how a provider says what it cannot do + +Not every provider implements every optional part of the contract. A scenario exercising one carries +a tag; a provider declares the tags it supports, and a scenario gated on an undeclared tag is reported +**skipped, with the reason** — never as passed. A suite that quietly goes green on scenarios it did +not run is worse than no suite at all, so this rule is the centre of the design rather than a detail. + +Tags compose: a scenario carrying two tags runs only if both are declared. + +| Tag | Meaning | +| --- | --- | +| `@events` | emits lifecycle events at all | +| `@lifecycle` | performs an initialisation that reaches its backend, with an observable outcome | +| `@stale` | enters `STALE` and emits `PROVIDER_STALE` on backend loss | +| `@configuration-change` | detects configuration changes and emits `PROVIDER_CONFIGURATION_CHANGED` | +| `@object` | supports structured flag values | +| `@variants` | names the variant it resolved | +| `@disabled-flags` | resolves a flag disabled in the management system to the code default | +| `@unavailable` | reports an error state instead of hanging against a dead backend | +| `@numeric-coercion` | coerces between integer and float only when lossless, else `TYPE_MISMATCH` | +| `@string-typing` | reports `TYPE_MISMATCH` for a boolean or integer flag requested as a string, rather than its string representation | +| `@fully-typed-values` | records a native type for float and structured values too, so the same question can be asked of them | +| `@large-integers` | resolves integers up to 2^53 − 1 exactly | +| `@reinitialization` | can be initialised again after `shutdown` | +| `@targeting` | resolves a flag differently for a matching evaluation context | +| `@standard-reasons` | reports the standard resolution reasons | +| `@caching` | reserved; **not declarable** — no scenarios carry it yet | + +Untagged scenarios are mandatory and always run. + +**Withholding a tag is not an admission of a defect.** Some of these describe behaviour the +specification does not require — `@numeric-coercion` borrows its rule from flagd's coercion ADR, and +`@string-typing` and `@fully-typed-values` sit on a question the specification does not answer at all +([#433](https://github.com/open-feature/spec/issues/433), [#430](https://github.com/open-feature/spec/issues/430)). +A provider that withholds one of those is not violating the specification, and this suite must not be +read as saying it is. + +## These three travel together + +A feature file that evaluates `boolean-flag` is meaningless without the flag definition, and a disconnect scenario is meaningless without the control endpoint that produces the disconnect. Changing one without the others breaks the suite in every language at once. + +## Five properties that are load-bearing + +- **`missing-flag` must not exist** in the flag set. Its absence is what the `FLAG_NOT_FOUND` scenario tests. Seeding it turns that scenario green for the wrong reason. +- **Only `targeting-key-flag` has a targeting rule.** Every other enabled flag resolves to its default variant whatever the evaluation context, which is what lets the untargeted scenarios expect reason `STATIC`. Seeding targeting onto any other flag breaks them in every language at once. Its rule is specified by behaviour — resolve `hit` when the targeting key is exactly `5c3d8535-f81a-4478-a6d3-afaa4d51199e`, `miss` otherwise — so express it however your backend expresses targeting. The flag, its variants and the uuid are the ones [flagd-testbed's `targeting.feature`](https://github.com/open-feature/flagd-testbed) already uses, on the same reasoning as the zero flags: a backend serving that harness already serves this. +- **`boolean-zero-flag`, `integer-zero-flag` and `string-zero-flag` resolve to falsy values on purpose.** A seeding step that treats `false`, `0` or `""` as "unset" and drops them turns the falsy-value scenarios into `FLAG_NOT_FOUND` failures that look like provider defects. These names, and their `zero`/`non-zero` variants, are the ones [Appendix B's SDK suite](../gherkin/test-flags.json) already uses, so a backend serving that flag set already serves these. +- **The four `disabled-*` flags are the only ones whose state is not `ENABLED`.** They resolve to nothing — the caller's default stands in, and no variant is named. Every other scenario assumes a flag serves its own value, so enabling one of these, or disabling anything else, breaks that assumption silently. Names, variants and values are [flagd-testbed's own](https://github.com/open-feature/flagd-testbed), from `flags/disabled-flags.json`. +- **`integral-float-flag` is a float and `huge-integer-flag` is an integer.** Seeding `10.0` as `10` makes the lossless-coercion scenario pass without coercing; seeding `9007199254740991` through a float rounds it. + +The flag set is expressed in the flagd flag-definition format because that is the only widely implemented vendor-neutral format today. The format is not what matters — the keys, types, variant names and resolved values are. Seed them however your backend seeds flags. + +## Consuming from Go + +This directory is also a Go module, `github.com/open-feature/spec/specification/assets/provider-tck`, whose only content is an `embed.FS` of the artifacts above. The Go conformance suite depends on it instead of vendoring a copy: a Go module ships as a zip of the VCS tree, in which a git submodule is only a gitlink, so an embed from a submodule would arrive empty for anyone running `go get`. The other languages build from a working tree and keep using the submodule; `go.mod` and `embed.go` are inert for them. + +A consumer pins a release the usual way: + +```console +go get github.com/open-feature/spec/specification/assets/provider-tck@v0.1.0 +``` + +## Releases + +These assets are released independently of the specification, by [release-please](../../../release-please-config.json). A nested Go module is tagged with its path as a prefix, so a release is tagged `specification/assets/provider-tck/vX.Y.Z` — the same shape as `providers/flagd/v0.6.0` in the SDK contrib repositories. The specification's own `vX.Y.Z` tags are cut by release-please too, from a separate release pull request that excludes this directory, and do not apply here; nothing about the two numbering schemes is related. + +What a bump means is not the usual thing, because this is a test suite rather than a library: + +- **A minor bump may turn a passing suite red.** Adding a scenario, or tightening one, raises the bar a provider has to clear. Nothing changed on the adopter's side and their build can still go from green to red, which is the point of adopting a conformance suite and is why new scenarios are released as minors rather than as patches. +- **A patch bump cannot.** Patches are editorial: a clarified scenario name, a comment, a fix to something that never ran. + +So pinning is not optional bookkeeping. A suite that floats on the latest assets cannot distinguish a regression in the provider from a new question being asked of it. + +## Consuming from the other three languages + +Java, Python and JavaScript reach these files through a git submodule of this repository, because a JAR, a wheel and an npm package are all built from a working tree where the submodule is present. A submodule can track the release tag rather than a bare commit, which makes the pin readable in review: + +```ini +[submodule "spec"] + path = tools/provider-tck/spec + url = https://github.com/open-feature/spec.git + branch = specification/assets/provider-tck/v0.1.0 +``` + +The recorded gitlink is still a commit, so `git submodule update --init` and `actions/checkout` with `submodules: recursive` behave exactly as before. Only `git submodule update --remote` is affected, which resolves `branch` and will report that the tag is not a branch — do not use it on a submodule pinned this way. + +## Keeping a pin current + +Both forms are updatable by [Renovate](https://docs.renovatebot.com), so an adopting repository is told about a new release rather than discovering it: + +- Go: the `gomod` manager, on by default, raises a PR for a new `specification/assets/provider-tck/vX.Y.Z`. +- The submodule: the `git-submodules` manager, which is opt-in and reads the tag out of `branch` above. + +```json +{ + "git-submodules": { "enabled": true } +} +``` + +Let those PRs run the suite. A red one is the report that conformance narrowed, and reading it is the work — which is why it is worth *not* automerging these. diff --git a/specification/assets/provider-tck/embed.go b/specification/assets/provider-tck/embed.go new file mode 100644 index 00000000..926e0f75 --- /dev/null +++ b/specification/assets/provider-tck/embed.go @@ -0,0 +1,35 @@ +// Package providertck carries the provider conformance assets as a Go module, +// so a Go conformance suite can depend on a specific revision of them the way +// it depends on any other module. +// +// The other language suites consume this directory through a git submodule, +// which works because a wheel or a JAR is built from a working tree where the +// submodule is present. A Go module is distributed as a zip built from the +// VCS tree, where a submodule is only a gitlink and its files are absent, so +// the Go suite would otherwise have to commit a copy of every artifact and +// police it against drift. Publishing the artifacts as a module removes the +// copy: the consumer pins a commit or tag in its go.mod, the Go checksum +// database makes that revision immutable, and the embedded bytes are the same +// bytes every other language reads out of the submodule. +// +// The module contains no code beyond this file and has no dependencies. It is +// inert for every consumer that is not Go. See +// https://github.com/open-feature/spec/issues/417. +package providertck + +import "embed" + +// FS holds the conformance artifacts, keyed by their path relative to this +// directory, so that they are addressed here exactly as they are documented: +// +// gherkin/*.feature the canonical scenarios +// flags/canonical-flags.json the flag set those scenarios assume +// openapi/control-api.yaml the HTTP surface a backend under test exposes +// +// The README and .gitattributes beside them are not artifacts and are not +// embedded. +// +//go:embed gherkin/*.feature +//go:embed flags/canonical-flags.json +//go:embed openapi/control-api.yaml +var FS embed.FS diff --git a/specification/assets/provider-tck/flags/canonical-flags.json b/specification/assets/provider-tck/flags/canonical-flags.json new file mode 100644 index 00000000..fec6e6ec --- /dev/null +++ b/specification/assets/provider-tck/flags/canonical-flags.json @@ -0,0 +1,227 @@ +{ + "$comment": [ + "The canonical flag set the TCK's feature files assume. A backend under test MUST serve an", + "equivalent set under the configuration named 'default'.", + "", + "Expressed in the flagd flag-definition format because that is the only widely implemented", + "vendor-neutral format today. The format is not what matters — the keys, types, variant", + "names and resolved values are. Seed them however your backend seeds flags.", + "", + "Two things are load-bearing and easy to get wrong:", + " * 'missing-flag' MUST NOT exist. Its absence is what the FLAG_NOT_FOUND scenario tests.", + " * Only 'targeting-key-flag' has a targeting rule. Everything else resolves the same way", + " whatever the context, because the TCK tests the provider's mapping of a response, not", + " the backend's evaluation logic. That is also what lets a provider declaring", + " @standard-reasons expect STATIC rather than TARGETING_MATCH for those flags.", + "", + "Three more are easy to lose in translation, because a seeding step that 'cleans up' values", + "destroys exactly what they test:", + " * 'boolean-zero-flag', 'integer-zero-flag' and 'string-zero-flag' resolve to false, 0 and", + " \"\". They are values, not absences. A backend that drops them, or a provider that", + " treats them as missing, is what the falsy-value scenarios catch.", + " * 'large-integer-flag' and 'huge-integer-flag' resolve to 2147483647 (2^31 - 1) and", + " 9007199254740991 (2^53 - 1). Seed both as integers, not floats, and do not round them: a", + " float32 round trip changes the first, and a 32-bit integer cannot hold the second.", + " * 'integral-float-flag' resolves to 10.0, a float with no fractional part. It MUST be seeded", + " as a float. A backend that stores it as the integer 10 makes the lossless-coercion scenario", + " pass without coercing anything." + ], + "flags": { + "boolean-flag": { + "state": "ENABLED", + "variants": { + "on": true, + "off": false + }, + "defaultVariant": "on" + }, + "string-flag": { + "state": "ENABLED", + "variants": { + "greeting": "hi", + "parting": "bye" + }, + "defaultVariant": "greeting" + }, + "integer-flag": { + "state": "ENABLED", + "variants": { + "one": 1, + "ten": 10 + }, + "defaultVariant": "ten" + }, + "float-flag": { + "state": "ENABLED", + "variants": { + "tenth": 0.1, + "half": 0.5 + }, + "defaultVariant": "half" + }, + "large-integer-flag": { + "$comment": "2^31 - 1, the largest 32-bit signed integer. Every language represents it exactly; a float32 round trip does not.", + "state": "ENABLED", + "variants": { + "one": 1, + "max-int32": 2147483647 + }, + "defaultVariant": "max-int32" + }, + "huge-integer-flag": { + "$comment": "2^53 - 1, the largest integer JavaScript represents exactly. Only asked for under @large-integers, because a 32-bit integer accessor cannot request it at all.", + "state": "ENABLED", + "variants": { + "one": 1, + "max-safe": 9007199254740991 + }, + "defaultVariant": "max-safe" + }, + "integral-float-flag": { + "$comment": "A float with no fractional part, for the lossless half of @numeric-coercion. Seed as a float.", + "state": "ENABLED", + "variants": { + "tenth": 0.1, + "ten": 10.0 + }, + "defaultVariant": "ten" + }, + "boolean-zero-flag": { + "$comment": "Resolves to false. The default in the scenario is true, so a provider that treats false as missing is caught.", + "state": "ENABLED", + "variants": { + "zero": false, + "non-zero": true + }, + "defaultVariant": "zero" + }, + "integer-zero-flag": { + "$comment": "Resolves to 0. The default in the scenario is 1, so a provider that treats 0 as missing is caught.", + "state": "ENABLED", + "variants": { + "zero": 0, + "non-zero": 1 + }, + "defaultVariant": "zero" + }, + "string-zero-flag": { + "$comment": "Resolves to the empty string. The default in the scenario is 'fallback', so a provider that treats \"\" as missing is caught.", + "state": "ENABLED", + "variants": { + "zero": "", + "non-zero": "str" + }, + "defaultVariant": "zero" + }, + "object-flag": { + "state": "ENABLED", + "variants": { + "empty": {}, + "template": { + "showImages": true, + "title": "Check out these pics!", + "imagesPerPage": 100 + } + }, + "defaultVariant": "template" + }, + "wrong-flag": { + "$comment": "A string flag, evaluated as a boolean by the TYPE_MISMATCH scenario.", + "state": "ENABLED", + "variants": { + "one": "uno", + "two": "dos" + }, + "defaultVariant": "one" + }, + "disabled-boolean-flag": { + "$comment": [ + "The four disabled flags mirror boolean-flag, string-flag, integer-flag and float-flag", + "exactly, differing only in state. Each scenario's caller default differs from the", + "flag's configured value, so a provider that ignores the state returns the configured", + "value and is caught on the value alone.", + "", + "Names, variants and values are flagd-testbed's own, from flags/disabled-flags.json,", + "so a backend serving that harness already serves these.", + "", + "There is deliberately no disabled-object-flag. An Object resolution needs @object, and", + "a scenario needing both tags cannot be one row of a single outline -- the scalar rows", + "already establish the behaviour." + ], + "state": "DISABLED", + "variants": { + "on": true, + "off": false + }, + "defaultVariant": "on" + }, + "disabled-string-flag": { + "state": "DISABLED", + "variants": { + "greeting": "hi", + "parting": "bye" + }, + "defaultVariant": "greeting" + }, + "disabled-integer-flag": { + "state": "DISABLED", + "variants": { + "one": 1, + "ten": 10 + }, + "defaultVariant": "ten" + }, + "disabled-float-flag": { + "state": "DISABLED", + "variants": { + "tenth": 0.1, + "half": 0.5 + }, + "defaultVariant": "half" + }, + "targeting-key-flag": { + "$comment": [ + "The one flag in this set with a targeting rule. Everything else resolves to its", + "defaultVariant whatever the context, which is what lets a provider declaring", + "@standard-reasons expect STATIC for those flags and TARGETING_MATCH or DEFAULT here.", + "", + "The rule is specified by its behaviour, not by this encoding: resolve variant 'hit'", + "when the evaluation context's targeting key is exactly the uuid below, and 'miss'", + "otherwise. Express that however your backend expresses targeting.", + "", + "Name, variants and uuid are the ones flagd-testbed's own targeting.feature already", + "uses, on the same reasoning as the zero flags: a backend serving that harness already", + "serves this, so adopting the canonical set costs it nothing.", + "", + "It is what makes context passthrough observable. A matching context resolves to a", + "different value, so a provider that drops the context on the floor is caught by the", + "resolved value itself rather than needing an echo endpoint." + ], + "state": "ENABLED", + "variants": { + "miss": "miss", + "hit": "hit" + }, + "defaultVariant": "miss", + "targeting": { + "if": [ + { "==": [{ "var": "targetingKey" }, "5c3d8535-f81a-4478-a6d3-afaa4d51199e"] }, + "hit", + null + ] + } + }, + "changing-flag": { + "$comment": [ + "The flag POST /change mutates. The TCK asserts only that its resolved value differs", + "after the change, so which of the two variants you start from does not matter." + ], + "state": "ENABLED", + "variants": { + "foo": "foo", + "bar": "bar" + }, + "defaultVariant": "foo" + } + } +} diff --git a/specification/assets/provider-tck/gherkin/errors.feature b/specification/assets/provider-tck/gherkin/errors.feature new file mode 100644 index 00000000..6d184d86 --- /dev/null +++ b/specification/assets/provider-tck/gherkin/errors.feature @@ -0,0 +1,143 @@ +Feature: Provider error handling + + # Every scenario here asserts the same three-part contract, because all three parts matter and + # providers routinely get one of them wrong: + # + # 1. the code default is returned — an application must keep working, + # 2. the correct error code is reported — an application must be able to tell what went wrong, + # 3. nothing is thrown — an unhandled exception from a flag evaluation is never acceptable. + # + # Requires the backend to be seeded with the canonical flag set — see flags/canonical-flags.json. + + Background: + Given a stable provider + + Scenario Outline: Requesting the wrong type returns the code default + # The requests no representation can satisfy honestly. Two questions are held out of this + # matrix because they have defensible answers rather than obvious ones: whether a number fits + # a narrower accessor (@numeric-coercion) and whether any flag may be returned through the + # string accessor (@string-typing). What is left is the set no backend can satisfy -- "hello" + # is not a boolean, and true is not a number, however the backend stores them. + Given a -flag with key "" and a default value "" + When the flag was evaluated with details + Then the resolved details value should be "" + And the error-code should be "TYPE_MISMATCH" + And no exception should have been thrown + + Examples: a string flag requested as something else + | key | requested | default | + | string-flag | Boolean | false | + | string-flag | Integer | 1 | + | string-flag | Float | 0.1 | + | wrong-flag | Boolean | false | + + Examples: a boolean flag requested as something else + | key | requested | default | + | boolean-flag | Integer | 1 | + | boolean-flag | Float | 0.1 | + + Examples: a numeric flag requested as a non-numeric type + | key | requested | default | + | integer-flag | Boolean | false | + | float-flag | Boolean | false | + + @object + Scenario Outline: Requesting a structured flag as a scalar returns the code default + Given a -flag with key "object-flag" and a default value "" + When the flag was evaluated with details + Then the resolved details value should be "" + And the error-code should be "TYPE_MISMATCH" + And no exception should have been thrown + + Examples: + | requested | default | + | Boolean | false | + | Integer | 1 | + | Float | 0.1 | + + @numeric-coercion + Scenario: A float flag is not silently narrowed to an integer + # 'float-flag' resolves to 0.5. Narrowing that to an integer would lose information + # silently, so it must be reported as a type mismatch rather than rounded. + # + # This is the lossy half of the coercion contract; the two scenarios that follow are the + # lossless half. A provider declaring @numeric-coercion must satisfy all three. Rejecting + # 0.5 is easy to get right by rejecting every float, and the lossless scenarios are what + # stop that shortcut from passing. + Given a Integer-flag with key "float-flag" and a default value "1" + When the flag was evaluated with details + Then the resolved details value should be "1" + And the error-code should be "TYPE_MISMATCH" + And no exception should have been thrown + + @numeric-coercion + Scenario: An integral float requested as an integer is coerced without loss + # 'integral-float-flag' resolves to 10.0. Nothing is lost by returning it as the integer + # 10, so the coercion rule permits it and a provider declaring the tag must perform it. + Given a Integer-flag with key "integral-float-flag" and a default value "1" + When the flag was evaluated with details + Then the resolved details value should be "10" + And the error-code should be "" + And no exception should have been thrown + + @numeric-coercion + Scenario: An integer requested as a float is widened without loss + # The other direction. 'integer-flag' resolves to 10; every integer this suite asks for + # is exactly representable as a float, so a provider declaring the tag must widen it. + Given a Float-flag with key "integer-flag" and a default value "0.1" + When the flag was evaluated with details + Then the resolved details value should be "10" + And the error-code should be "" + And no exception should have been thrown + + @string-typing + Scenario Outline: A non-string flag is not returned as its string representation + # Every value has a string representation, so a backend that stores flag values as strings + # satisfies the string accessor for every flag and has no mismatch to report. Its flags are + # strings, and Requirement 2.2.3 asks it for the resolved flag value, which it returned. + # Whether that is wrong is not something this suite can assert -- see the appendix. + # + # These two rows are the ones a partially typed backend can still answer: a boolean and an + # integer are types such a store records natively. The float and structured cases are held + # separately behind @fully-typed-values, because a backend can lack a type for those while + # having one for these -- and one tag covering both would report a provider that fails these + # as merely untyped. + Given a String-flag with key "" and a default value "fallback" + When the flag was evaluated with details + Then the resolved details value should be "fallback" + And the error-code should be "TYPE_MISMATCH" + And no exception should have been thrown + + Examples: + | key | + | boolean-flag | + | integer-flag | + + @string-typing @fully-typed-values + Scenario: A float flag is not returned as its string representation + # Held apart from the two rows above because a store can record booleans and integers + # natively and still keep floats as text, which is what @fully-typed-values asks about. + Given a String-flag with key "float-flag" and a default value "fallback" + When the flag was evaluated with details + Then the resolved details value should be "fallback" + And the error-code should be "TYPE_MISMATCH" + And no exception should have been thrown + + @object @string-typing @fully-typed-values + Scenario: A structured flag is not returned as its JSON text + # The same property one type further out. Three tags: a provider with no structured values + # cannot be asked at all (@object), and a store that keeps structures as text has nothing + # to report a mismatch about (@fully-typed-values). + Given a String-flag with key "object-flag" and a default value "fallback" + When the flag was evaluated with details + Then the resolved details value should be "fallback" + And the error-code should be "TYPE_MISMATCH" + And no exception should have been thrown + + Scenario: An unknown flag key returns the code default + # 'missing-flag' is deliberately absent from the canonical flag set. + Given a String-flag with key "missing-flag" and a default value "fallback" + When the flag was evaluated with details + Then the resolved details value should be "fallback" + And the error-code should be "FLAG_NOT_FOUND" + And no exception should have been thrown diff --git a/specification/assets/provider-tck/gherkin/evaluation.feature b/specification/assets/provider-tck/gherkin/evaluation.feature new file mode 100644 index 00000000..6d1a0b59 --- /dev/null +++ b/specification/assets/provider-tck/gherkin/evaluation.feature @@ -0,0 +1,258 @@ +Feature: Provider flag evaluation + + # Verifies that a provider maps backend responses onto typed resolution details correctly. + # + # This does NOT test the backend's evaluation logic. Every flag in the canonical set except + # targeting-key-flag resolves to its default variant whatever the context, so what is under + # test is purely the provider's mapping of a backend response to a value and a variant. + # targeting-key-flag carries the one rule, only to show the context reached the backend. + # + # No scenario here asserts a resolution reason. The reasons are a provider's claim to use the + # standard vocabulary, checked in reason.feature behind @standard-reasons. + # + # Every success path also asserts that no error message was set (requirement 2.3.2). A + # provider that reports a value AND an error message is sending two contradictory signals, + # and an application reading the message will believe the wrong one. + # + # Requires the backend to be seeded with the canonical flag set — see flags/canonical-flags.json. + + Background: + Given a stable provider + + Scenario Outline: Resolve values + Given a -flag with key "" and a default value "" + When the flag was evaluated with details + Then the resolved details value should be "" + And the error-code should be "" + And the error message should be empty + And no exception should have been thrown + + Examples: + | key | type | default | value | + | boolean-flag | Boolean | false | true | + | string-flag | String | bye | hi | + | integer-flag | Integer | 1 | 10 | + | float-flag | Float | 0.1 | 0.5 | + + Scenario Outline: A falsy value is a value, not an absence + # false, 0 and "" are the values most likely to be mistaken for "nothing came back": a + # `value || default` in JavaScript, a zero-value check in Go, an `if not value` in Python. + # Each row's default differs from its resolved value, so a provider that falls back on a + # falsy result returns the wrong value and is caught by the value assertion alone. + Given a -flag with key "" and a default value "" + When the flag was evaluated with details + Then the resolved details value should be "" + And the error-code should be "" + And the error message should be empty + And no exception should have been thrown + + Examples: + | key | type | default | value | + | boolean-zero-flag | Boolean | true | false | + | integer-zero-flag | Integer | 1 | 0 | + | string-zero-flag | String | fallback | | + + @variants + Scenario Outline: The resolved details name the variant + # Gated, because a variant is optional rather than required. types.md declares the field + # "variant (string, optional)", and Requirement 2.2.4 is a SHOULD: in normal execution a + # provider "SHOULD populate the resolution details structure's variant field". The same + # section goes further and says the value "might only be meaningful in the context of the + # flag management system associated with the provider". + # + # Some systems have no variant concept for a plain flag at all. Their evaluation response + # carries no such key, so the provider never receives one and no amount of seeding can + # produce one. Asserting a variant in every scenario failed such a backend ten times over + # for something that is not a defect and that no provider author can fix — and left nothing + # to record as a known deviation, because there was no capability to hang one on. + # + # A provider whose backend names its variants declares this tag and these rows run. One + # whose backend does not leaves it undeclared, and they are skipped with that reason rather + # than passed. Either way the value assertions above are unaffected: they are untagged, and + # 2.2.3 makes the value a MUST. + Given a -flag with key "" and a default value "" + When the flag was evaluated with details + Then the variant should be "" + And the error-code should be "" + And the error message should be empty + And no exception should have been thrown + + Examples: + | key | type | default | variant | + | boolean-flag | Boolean | false | on | + | string-flag | String | bye | greeting | + | integer-flag | Integer | 1 | ten | + | float-flag | Float | 0.1 | half | + | boolean-zero-flag | Boolean | true | zero | + | integer-zero-flag | Integer | 1 | zero | + | string-zero-flag | String | fallback | zero | + | large-integer-flag | Integer | 1 | max-int32 | + + Scenario: A large integer resolves without loss of precision + # 2147483647 is 2^31 - 1, the largest 32-bit signed integer, so every language's integer + # accessor can ask for it. It is also outside what a 32-bit float represents exactly, so a + # provider that routes integers through float32 and back returns 2147483648. + Given a Integer-flag with key "large-integer-flag" and a default value "1" + When the flag was evaluated with details + Then the resolved details value should be "2147483647" + And the error-code should be "" + And no exception should have been thrown + + @large-integers + Scenario: An integer beyond 32 bits resolves without loss of precision + # 9007199254740991 is 2^53 - 1: the largest integer JavaScript represents exactly, and + # comfortably inside a 64-bit integer. Anything that routes the value through a 32-bit + # integer, or through a 64-bit float and back with rounding, changes it. + # + # Tagged, because whether it can be asked for at all is a property of the language's + # SDK rather than of the provider: Java's integer accessor is a 32-bit Integer, and a + # provider cannot resolve a value the accessor has no room for. Values above 2^53 - 1 are + # deliberately not asked for. What a provider must do with a value that does not fit the + # requested accessor is an open question of the provider contract (open-feature/spec#430). + Given a Integer-flag with key "huge-integer-flag" and a default value "1" + When the flag was evaluated with details + Then the resolved details value should be "9007199254740991" + And the error-code should be "" + And no exception should have been thrown + + Scenario: An integer flag resolves as an integer + # Paired with the float scenario below and with the narrowing scenario in errors.feature. + # Together they pin down that the two numeric types stay distinct rather than both being + # funnelled through one numeric representation. + Given a Integer-flag with key "integer-flag" and a default value "1" + When the flag was evaluated with details + Then the resolved details value should be "10" + And the error-code should be "" + And the error message should be empty + And no exception should have been thrown + + Scenario: A float flag resolves as a float + Given a Float-flag with key "float-flag" and a default value "0.1" + When the flag was evaluated with details + Then the resolved details value should be "0.5" + And the error-code should be "" + And the error message should be empty + And no exception should have been thrown + + @disabled-flags + Scenario Outline: A disabled flag resolves to the code default + # Gated, because it takes a provider and its backend together. The backend has to + # distinguish a disabled flag at all — flagd says so with reason DISABLED and no variant, + # and OFREP's codeDefaultFlag schema carries the same signal by omitting `value` — and the + # provider then has to substitute the caller's default on the strength of that signal. A + # backend that instead serves the flag's configured value leaves nothing to detect, and a + # provider that does not substitute cannot pass however clear the signal was. + # + # Where evaluation happens is not the axis. flagd's RPC resolver is remote and passes, by + # substituting locally when the reason is DISABLED and no variant came back; an OFREP + # provider can and does pass on exactly the same reasoning. + # + # Nothing in the specification says what a provider owes a disabled flag. Requirement + # 1.4.7 is about the SDK propagating whatever reason arrived, and 2.2.5 only lists + # DISABLED among the reason strings a provider may use. So this appendix states the + # behaviour, the way it does for @numeric-coercion, and gates it. + # + # Asserts the value and the absence of an error, not the reason. Each row's caller default + # differs from the flag's configured value, so a provider that ignores the state returns + # the configured value and fails on the value alone — which rests on 2.2.3, a MUST. + # Pinning reason "DISABLED" here would rest on 2.2.5, a SHOULD that permits "some other + # string"; it is pinned in reason.feature instead, for providers that declare + # @standard-reasons and so opt into the standard meanings. + # + # No variant is asserted. A disabled flag has resolved no variant, so there is none to + # name; this scenario and @variants deliberately do not compose. + Given a -flag with key "" and a default value "" + When the flag was evaluated with details + Then the resolved details value should be "" + And the error-code should be "" + And the error message should be empty + And no exception should have been thrown + + Examples: + | key | type | default | + | disabled-boolean-flag | Boolean | false | + | disabled-string-flag | String | bye | + | disabled-integer-flag | Integer | 1 | + | disabled-float-flag | Float | 0.1 | + + @object + Scenario: Resolve a structured value + Given a Object-flag with key "object-flag" and a default value "{}" + When the flag was evaluated with details + Then the error-code should be "" + And the error message should be empty + And no exception should have been thrown + And the resolved object value should contain + | key | type | value | + | showImages | Boolean | true | + | title | String | Check out these pics! | + | imagesPerPage | Integer | 100 | + Scenario: Supplying an evaluation context does not disturb an untargeted resolution + # Mandatory, and the only scenario that passes a context to a provider with no targeting + # involved. Requirement 2.2.1 makes the evaluation context a parameter of every resolve + # method, but until this scenario existed no scenario supplied one — so a provider that + # threw on any context, or serialised it into a malformed request, passed the whole suite. + # + # string-flag has no targeting rule, so the context cannot change the outcome. What is + # under test is only that supplying one is harmless. + # + # The step wording is Appendix B's, and the Java TCK already carries a step definition for + # it, because inventing a second way to say "a context containing a targeting key" is the + # kind of divergence this appendix exists to prevent. + # + # Deliberately asserts the value and the absence of an error rather than the reason. + # 2.2.3 makes the value a MUST and 2.2.6 forbids an error code in normal execution, while + # the reason is a SHOULD that 2.2.5 lets a provider populate with "some other string". + # Reasons are asserted only in reason.feature, behind @standard-reasons. + Given a String-flag with key "string-flag" and a default value "bye" + And a context containing a targeting key with value "f20bd32d-703b-48b6-bc8e-79d53c85134a" + When the flag was evaluated with details + Then the resolved details value should be "hi" + And the error-code should be "" + And the error message should be empty + And no exception should have been thrown + + @targeting + Scenario: A matching evaluation context resolves the targeted variant + # This is what makes context passthrough observable. Every other flag resolves the same way + # whatever the context, so a provider that drops the context entirely passes them all. Here + # a matching context resolves to a different value, so dropping it is caught by the resolved + # value itself — no echo endpoint on the control API required. + # + # targeting-key-flag's rule is specified by behaviour, not by syntax: resolve "hit" when the + # targeting key is exactly this uuid, "miss" otherwise. Express it however your backend + # expresses targeting. The flag, its variants and the uuid are flagd-testbed's own, so a + # backend serving that harness already serves this one. + # + # Asserts the value, not the reason: the resolved value carries the whole signal, and a + # provider that reports vendor-specific reasons is conformant. TARGETING_MATCH for this hit + # and DEFAULT for the miss below are asserted in reason.feature, which composes + # @standard-reasons with @targeting so that both must be declared. + Given a String-flag with key "targeting-key-flag" and a default value "fallback" + And a context containing a targeting key with value "5c3d8535-f81a-4478-a6d3-afaa4d51199e" + When the flag was evaluated with details + Then the resolved details value should be "hit" + And the error-code should be "" + And no exception should have been thrown + + @targeting + Scenario: A non-matching evaluation context resolves the default variant + # Paired with the scenario above, and the reason it is not enough on its own: a provider + # that always returned the targeted value would pass that one. This is what pins down that + # the rule was evaluated rather than the targeted variant simply being served. + Given a String-flag with key "targeting-key-flag" and a default value "fallback" + And a context containing a targeting key with value "f20bd32d-703b-48b6-bc8e-79d53c85134a" + When the flag was evaluated with details + Then the resolved details value should be "miss" + And the error-code should be "" + And no exception should have been thrown + + @targeting + Scenario: No evaluation context resolves the default variant + # A targeting rule that cannot match must not error. A provider that requires a targeting + # key, or that fails to evaluate a rule when the context is absent, is caught here. + Given a String-flag with key "targeting-key-flag" and a default value "fallback" + When the flag was evaluated with details + Then the resolved details value should be "miss" + And the error-code should be "" + And no exception should have been thrown diff --git a/specification/assets/provider-tck/gherkin/events.feature b/specification/assets/provider-tck/gherkin/events.feature new file mode 100644 index 00000000..00e7e5ef --- /dev/null +++ b/specification/assets/provider-tck/gherkin/events.feature @@ -0,0 +1,42 @@ +@events +Feature: Provider events + + # Verifies that a provider notices changes in its backend and both signals them and acts on + # them. Signalling alone is not enough: a configuration-change event that is not followed by + # a changed evaluation result is a lie, so each scenario asserts the event AND the behaviour. + # + # Outages here are simulated inside the running stack via the control API. No container is + # ever stopped or restarted — see the invariant in openapi/control-api.yaml. + + Background: + Given a stable provider + + @configuration-change + Scenario: A configuration change is signalled and applied + Given a String-flag with key "changing-flag" and a default value "unset" + And a change event handler + When the flag was evaluated with details + And the resolved value is remembered + And the flag was modified + Then the change event handler should have been executed + And the flag should be part of the event payload + When the flag was evaluated with details + Then the resolved details value should have changed + And no exception should have been thrown + + @stale + Scenario: Losing the backend makes the provider stale, regaining it makes it ready again + Given a ready event handler + And a stale event handler + When a ready event was fired + And the connection is lost + Then the stale event handler should have been executed + And the client should be in stale state + When the connection is restored + Then the ready event handler should have been executed + And the client should be in ready state + + # Deliberately NOT covered here: whether a stale provider keeps serving last-known values + # during the outage. That is caching behaviour, which depends on whether the provider holds a + # local copy of the ruleset, and it belongs behind the @caching capability once those + # scenarios are written. See the "Known gaps" section of the README. diff --git a/specification/assets/provider-tck/gherkin/lifecycle.feature b/specification/assets/provider-tck/gherkin/lifecycle.feature new file mode 100644 index 00000000..16033456 --- /dev/null +++ b/specification/assets/provider-tck/gherkin/lifecycle.feature @@ -0,0 +1,91 @@ +@lifecycle +Feature: Provider lifecycle + + # Verifies the two terminal outcomes of provider initialisation: reaching READY against a + # healthy backend, and settling into ERROR against one that cannot be reached. And the other + # end of the lifecycle: that shutdown releases what initialisation acquired, can be repeated, + # and does not hang when the backend is gone. + # + # Gated by @lifecycle rather than @events, and the distinction is load-bearing. Every SDK + # synthesises PROVIDER_READY for a provider that has no initialisation step, so a provider + # without a lifecycle passes the readiness scenario below without demonstrating anything -- + # a NoOpProvider passes it identically. @lifecycle asserts that the provider actually reaches + # its backend during initialisation and that the outcome is observable; a provider that merely + # emits events does not necessarily do that. + # + # The failure case matters more than it looks. A provider that blocks forever, or throws out + # of provider registration, takes the host application down with it -- so the requirement is + # not merely that initialisation fails, but that it fails observably and promptly. + # + # The shutdown steps call the provider's own shutdown and initialize functions directly, not + # the SDK's. The SDK shuts a provider down when it is replaced, but wrapping that in a scenario + # would test the SDK's bookkeeping as much as the provider's, and Appendix B already does that. + + Scenario: A provider that successfully initializes becomes ready + Given a stable provider + And a ready event handler + Then the ready event handler should have been executed + And the client should be in ready state + + @unavailable + Scenario: A provider that cannot reach its backend reports an error + Given a unavailable provider + And a error event handler + Then the error event handler should have been executed within 10000ms + And the client should be in error state + + @unavailable + Scenario: A provider that cannot reach its backend still returns code defaults + Given a unavailable provider + And a error event handler + And a Boolean-flag with key "boolean-flag" and a default value "false" + Then the error event handler should have been executed within 10000ms + When the flag was evaluated with details + Then the resolved details value should be "false" + And no exception should have been thrown + + Scenario: Shutting down a provider twice has no further effect + # Requirement 2.5.3. Double-close is the classic shutdown bug: the second call finds a + # closed channel, a null connection or a disposed client, and throws from a code path the + # application runs during its own shutdown -- where an exception is least welcome. + Given a stable provider + When the provider is shut down + And the provider is shut down + Then no exception should have been thrown + + @reinitialization + Scenario: A provider that was shut down can be initialized again + # Requirement 2.5.2 says a provider SHOULD revert to its uninitialized state after shutdown, + # and its supporting text says "some providers MAY allow reinitialization from this state". + # Reuse is therefore permitted, not required, and this scenario is gated accordingly: a + # provider that shuts down by discarding its client and never recreating it is making a + # choice the specification allows, not exhibiting a defect. + # + # What the tag buys is the other direction. A provider that does claim to be reusable has + # somewhere to be held to it, because "shutdown() releases the client and initialize() + # returns early because an initialised flag was never cleared" is easy to write and leaves + # the provider evaluating against a closed connection rather than failing outright. + # + # Reverting the state itself is not separately observable: a provider that reverts but + # refuses reuse presents exactly as one that did neither. So this is the only assertion the + # requirement admits, and it only applies where reuse is offered. + Given a stable provider + And a Boolean-flag with key "boolean-flag" and a default value "false" + When the provider is shut down + And the provider is initialized again + And the flag was evaluated with details + Then the resolved details value should be "true" + And the error-code should be "" + And no exception should have been thrown + + @unavailable + Scenario: Shutting down a provider that cannot reach its backend completes promptly + # A shutdown that waits for a graceful close of a connection that will never answer hangs + # the host application's own shutdown. The bound is generous; what is being asserted is + # that shutdown returns at all rather than blocking on the backend. + Given a unavailable provider + And a error event handler + Then the error event handler should have been executed within 10000ms + When the provider is shut down + Then the shutdown should have completed within 10000ms + And no exception should have been thrown diff --git a/specification/assets/provider-tck/gherkin/metadata.feature b/specification/assets/provider-tck/gherkin/metadata.feature new file mode 100644 index 00000000..b8422447 --- /dev/null +++ b/specification/assets/provider-tck/gherkin/metadata.feature @@ -0,0 +1,9 @@ +Feature: Provider metadata + + # Verifies that a provider identifies itself (requirement 2.1.1). This looks too small to + # test, and was left untested for exactly that reason -- until a conformance report keyed on + # the provider's metadata name made an empty name into a report nobody can attribute. + + Scenario: A provider identifies itself by name + Given a stable provider + Then the provider metadata name should not be empty diff --git a/specification/assets/provider-tck/gherkin/reason.feature b/specification/assets/provider-tck/gherkin/reason.feature new file mode 100644 index 00000000..97bbc7da --- /dev/null +++ b/specification/assets/provider-tck/gherkin/reason.feature @@ -0,0 +1,120 @@ +@standard-reasons +Feature: Provider resolution reasons + + # Verifies that a provider reports the standard resolution reasons, with the meanings Appendix F + # gives them. + # + # THE WHOLE FILE IS GATED, and the gate is a claim rather than an excuse. Requirement 2.2.5 is a + # SHOULD, and it goes further than the other SHOULDs: a provider may populate `reason` with one of + # the listed values "or some other string indicating the semantic reason for the returned flag + # value". A provider whose backend reports vendor-specific reasons is therefore conformant, and + # asserting an exact reason against it would fail it for something the specification permits. + # + # So `@standard-reasons` is a provider saying "I use the standard vocabulary, with the standard + # meanings" — and this file is what checks that claim. A provider that does not say it leaves these + # scenarios skipped with their reason and loses nothing; its values and error codes are still + # asserted everywhere else, because those rest on MUSTs. The declaration is what a consumer reads + # when it wants to build telemetry, dashboards or debugging on `reason`: the claim was verified, + # not assumed. + # + # No other scenario in this suite asserts a reason. That is deliberate — an earlier revision + # asserted one in thirteen places across three files, which narrowed a SHOULD into a MUST for every + # adopter, and bought very little: every canonical flag resolves to a value distinct from the + # caller's default, so a provider that silently falls back is already caught by the value. + # + # TAGS COMPOSE. A scenario carrying `@targeting` or `@disabled-flags` needs that capability + # declared as well, because the reason cannot be observed without the behaviour that produces it. + # + # Two reasons are deliberately absent. CACHED belongs behind the reserved `@caching` tag and needs + # a repeat evaluation, which no scenario in this suite performs without a configuration change in + # between — see the caching note in Appendix F. STALE is reported during an outage, and what a + # provider serves while stale is itself uncovered, so there is nothing to attach it to yet. + # + # Requires the backend to be seeded with the canonical flag set — see flags/canonical-flags.json. + + Background: + Given a stable provider + + Scenario Outline: A flag with no targeting rules resolves statically + # This is the assertion the specification leaves genuinely open, and the reason this capability + # has to define its terms rather than merely name them. `types.md` types DEFAULT as "no dynamic + # evaluation occurred OR dynamic evaluation yielded no result", which a rule-less flag satisfies + # as readily as STATIC does. Appendix F picks STATIC for this capability — see its reason + # mapping — and a provider that answers DEFAULT here is not thereby defective, it simply does + # not use the standard meanings and should not declare the tag. + Given a -flag with key "" and a default value "" + When the flag was evaluated with details + Then the reason should be "STATIC" + And the error-code should be "" + And no exception should have been thrown + + Examples: + | key | type | default | + | boolean-flag | Boolean | false | + | string-flag | String | bye | + | integer-flag | Integer | 1 | + | float-flag | Float | 0.1 | + + # THE TWO ERROR SCENARIOS BELOW ASSERT AGREEMENT, not authorship, and the distinction is worth + # stating because it is the one place this file's subject is blurred. + # + # Every other scenario here rests on [Requirement 1.4.7](../../../sections/01-flag-evaluation.md), + # which makes the SDK propagate the provider's reason — but only "in cases of normal execution". + # Abnormal execution is 1.4.9, and that is a SHOULD on the *SDK* to "indicate an error"; nothing + # requires the provider's reason to survive. So a passing ERROR scenario does not establish that + # the provider set the reason, only that whatever reached the application is consistent. + # + # They are still worth running. The error code alone is already asserted in errors.feature, on a + # MUST, for every provider; the reason alone could be written by the SDK. Asserting the pair is + # the part neither field can satisfy on its own, and an evaluation that reports FLAG_NOT_FOUND + # with reason STATIC is incoherent whoever wrote it. + + Scenario: An unknown flag reports an error + Given a String-flag with key "missing-flag" and a default value "fallback" + When the flag was evaluated with details + Then the reason should be "ERROR" + And the error-code should be "FLAG_NOT_FOUND" + And no exception should have been thrown + + Scenario: A type mismatch reports an error + Given a String-flag with key "boolean-flag" and a default value "fallback" + When the flag was evaluated with details + Then the reason should be "ERROR" + And the error-code should be "TYPE_MISMATCH" + And no exception should have been thrown + + @targeting + Scenario: A matching targeting rule reports a targeting match + # Needs @targeting as well: a provider with no targeting has no rule to match, so there is no + # TARGETING_MATCH for it to report and the scenario would fail it for an absence rather than a + # defect. + Given a String-flag with key "targeting-key-flag" and a default value "fallback" + And a context containing a targeting key with value "5c3d8535-f81a-4478-a6d3-afaa4d51199e" + When the flag was evaluated with details + Then the reason should be "TARGETING_MATCH" + And the error-code should be "" + And no exception should have been thrown + + @targeting + Scenario: A targeting rule that does not match reports the default + # The other half, and the one that distinguishes the two reasons rather than merely observing + # one of them. A provider that reports TARGETING_MATCH whenever a rule exists, matched or not, + # passes the scenario above and fails this one. + Given a String-flag with key "targeting-key-flag" and a default value "fallback" + And a context containing a targeting key with value "f20bd32d-703b-48b6-bc8e-79d53c85134a" + When the flag was evaluated with details + Then the reason should be "DEFAULT" + And the error-code should be "" + And no exception should have been thrown + + @disabled-flags + Scenario: A disabled flag reports that it is disabled + # Needs @disabled-flags as well, for the same reason: a backend that cannot distinguish a + # disabled flag produces no DISABLED signal for the provider to pass on. The @disabled-flags + # scenarios themselves assert only the value, because pinning the reason there would have + # narrowed 2.2.5 for every adopter — this is where that narrowing is opted into instead. + Given a Boolean-flag with key "disabled-boolean-flag" and a default value "false" + When the flag was evaluated with details + Then the reason should be "DISABLED" + And the error-code should be "" + And no exception should have been thrown diff --git a/specification/assets/provider-tck/go.mod b/specification/assets/provider-tck/go.mod new file mode 100644 index 00000000..1080ae17 --- /dev/null +++ b/specification/assets/provider-tck/go.mod @@ -0,0 +1,3 @@ +module github.com/open-feature/spec/specification/assets/provider-tck + +go 1.21 diff --git a/specification/assets/provider-tck/openapi/control-api.yaml b/specification/assets/provider-tck/openapi/control-api.yaml new file mode 100644 index 00000000..e293876e --- /dev/null +++ b/specification/assets/provider-tck/openapi/control-api.yaml @@ -0,0 +1,429 @@ +openapi: 3.0.3 + +info: + title: OpenFeature Provider TCK — Backend Control API + version: 0.0.1 + description: | + The control API that a **backend under test** must expose so the OpenFeature + Provider TCK can drive it. + + The TCK verifies the *provider contract*: how a provider maps backend + responses to typed resolution details, lifecycle states and events. To do + that it must be able to put the backend into specific states on demand — + running, unreachable, reconfigured. This document standardises how. + + This specification is derived from the control endpoints already implemented + by [`flagd-testbed`](https://github.com/open-feature/flagd-testbed)'s + "launchpad" server, which is the reference implementation. + + ## Where this document should live + + This file currently ships inside the Java `provider-tck` artifact, but it is + not a Java artifact: it is a language-agnostic contract that every language's + TCK must implement identically, and that backend vendors implement in + whatever language their testbed is written in (Go, for flagd). + + It therefore belongs in the OpenFeature **spec** repository + (`open-feature/spec`), alongside the canonical Gherkin feature files and the + canonical flag set. Those three artifacts are a single unit — a feature file + that evaluates `boolean-flag` is meaningless without the flag definition, and + a disconnect scenario is meaningless without the endpoint that produces the + disconnect. Splitting them across repositories would let them drift. + + Each language's TCK then vendors the spec repo (git submodule or equivalent) + and packages these files into its own distribution format, so that adopting a + TCK never requires a consumer to check out a submodule of their own. + + ## Conformance language + + The key words MUST, MUST NOT, REQUIRED, SHOULD, SHOULD NOT and MAY are to be + interpreted as described in RFC 2119. + + Each operation below is tagged **REQUIRED** or **OPTIONAL**. A backend that + implements every REQUIRED operation can run the full TCK. OPTIONAL operations + have a defined fallback that the TCK applies automatically, so omitting them + costs nothing but precision. + + --- + + ## Normative requirement 1 — the no-container-restart invariant + + > **Container lifecycle operations MUST NOT be used to simulate backend + > unavailability. Backend unavailability MUST be simulated from inside the + > running stack.** + + The TCK starts the vendor's Docker Compose stack **once per test suite** and + reads the dynamically mapped host ports. Testcontainers cannot reliably + preserve mapped ports across a container stop/start in all language + bindings — a restarted container generally comes back on a *different* host + port, which silently invalidates every provider instance already pointed at + the old one. Any TCK implementation in any language hits this, so the + constraint is part of the contract rather than a Java detail. + + Therefore an implementation of `/stop`, `/restart` or any other outage + simulation MUST achieve the outage by one of: + + * killing or suspending the backend **process** inside its container + (the reference behaviour — this is what flagd-testbed does); + * a proxy in the stack refusing or blackholing connections + (e.g. a toxiproxy toxic, an envoy `direct_response`); + * an in-container firewall or socket-level block. + + An implementation MUST NOT `docker stop`, `docker kill`, `docker rm` or + recreate any container in the stack while the suite is running. The stack is + brought up before the first scenario and torn down after the last one, and + the mapped ports MUST remain stable for that entire window. + + --- + + ## Normative requirement 2 — flag state semantics across outages + + Outage simulation and flag-state seeding are orthogonal, and the TCK relies + on that separation for scenario isolation: + + * `POST /start` **MUST** (re)seed flag state to the baseline defined by the + named configuration. Any mutation previously applied by `POST /change` + MUST be discarded. This is what makes `/start` usable as a reset. + * `POST /restart` and a `POST /stop` followed by a `POST /start` **of the + same configuration** MUST leave the backend serving the same baseline + flag state it served before the outage. An outage MUST NOT be observable + as a change in flag *values* — only as a change in *availability*. + * `POST /change` mutations persist until the next `/start` or `/reset`. + + --- + + ## Normative requirement 3 — compose stack conventions + + The backend under test is delivered as a **Docker Compose stack**, not a + single image, so vendors can compose proxies, edge services or several + containers. The TCK only relies on these conventions: + + * One service — by default named `backend`, overridable by the provider + author — exposes the control API on container-internal port `8080` + (also overridable). + * The same stack exposes whatever port(s) the provider connects to. + * **All external ports are dynamically mapped.** A stack MUST NOT pin host + ports; the TCK discovers them after startup and hands them to the + provider factory. + * The stack MAY contain any number of additional services. + + --- + + ## Known gap — evaluation context passthrough + + The **targeting key** is covered: the `@targeting` scenarios resolve + `targeting-key-flag` with a matching and a non-matching context and get + different variants back, so a provider that drops the context entirely fails + them. + + What is still uncovered is **every other attribute**. A provider that forwards + the targeting key and silently discards the rest of the evaluation context + passes every scenario here. Closing that needs either an echo operation on + this API (e.g. `GET /last-evaluation`, returning the most recent request the + backend received) or a canonical flag whose rule keys on a custom attribute. + + license: + name: Apache 2.0 + url: https://www.apache.org/licenses/LICENSE-2.0 + +servers: + - url: http://{host}:{port} + description: | + Resolved at runtime from the Compose stack. `host` is the Docker host and + `port` is the dynamically mapped host port for the control service's + internal port 8080. + variables: + host: + default: localhost + port: + default: "8080" + +tags: + - name: lifecycle + description: Start and stop the backend process. + - name: availability + description: Simulate outages without touching containers. + - name: flags + description: Seed and mutate flag configuration. + - name: health + description: Readiness of the control API itself. + +paths: + + /start: + post: + tags: [lifecycle] + operationId: start + summary: "[REQUIRED] Start the backend and seed flags to a named baseline" + description: | + Starts the backend process using the named configuration and seeds flag + state to that configuration's baseline. + + MUST be idempotent in the sense that calling it while the backend is + already running is not an error: the implementation restarts the process + (or otherwise ensures it is running) with the requested configuration. + + **MUST NOT return until the seeded flag state is actually being served.** + A 200 is a promise that the very next evaluation will resolve against the + new baseline. Returning as soon as the process reports healthy is not + enough: a backend can accept connections and answer a readiness probe + while its flag store is still empty, and an evaluation in that window + gets `FLAG_NOT_FOUND` for a flag the configuration plainly defines. + + This is easy to get wrong and easy to miss. A provider that blocks during + initialisation -- streaming, or syncing a ruleset -- absorbs the window + and never sees it. A **stateless** provider, which evaluates over HTTP + with no initialisation at all, has nothing to hide it behind and fails + essentially every scenario, which reads as a catastrophically broken + provider rather than as a racing testbed. The reference implementation + exhibits this: its `/start` returns roughly 40ms before flagd's file + sources reach the flag store. + + A TCK MAY defensively probe after `/start`, but it should not have to, + and requiring every stateless adopter to reimplement that probe is worse + than stating the requirement here. + + Because this operation resets flag state, the TCK uses it as its default + scenario-isolation mechanism when `/reset` is not implemented. + + The set of valid configuration names is vendor-defined. Every + implementation MUST support the name `default`, which MUST serve the + canonical flag set the TCK's feature files assume. + + Reference implementation: flagd-testbed launches the `flagd` binary with + the config file of that name from `launchpad/configs` and rewrites + `/flags/allFlags.json`. + parameters: + - name: config + in: query + required: false + description: | + Name of the configuration to start with. Defaults to `default`. + schema: + type: string + default: default + example: default + responses: + "200": + description: Backend started and flag state seeded. + "400": + description: Unknown configuration name. + content: + application/json: + schema: + $ref: "#/components/schemas/Error" + + /stop: + post: + tags: [availability] + operationId: stop + summary: "[REQUIRED] Make the backend unreachable" + description: | + Makes the backend unreachable to the provider, simulating an outage. + + **MUST NOT stop the container.** See normative requirement 1. The + reference implementation kills the flagd process while its container + keeps running. + + The backend stays unreachable until a subsequent `POST /start`. Calling + `/stop` when the backend is already stopped MUST succeed. + + The TCK uses this to drive providers into `STALE` and `ERROR` states and + to assert `PROVIDER_STALE` / `PROVIDER_ERROR` events. + responses: + "200": + description: Backend is now unreachable; container still running. + + /restart: + post: + tags: [availability] + operationId: restart + summary: "[OPTIONAL] Simulate an outage of a bounded duration" + description: | + Makes the backend unreachable, waits `seconds`, then starts it again with + the configuration currently in effect. + + Flag state MUST be preserved across the outage — see normative + requirement 2. This is what distinguishes `/restart` from + `/stop` + `/start`: the former is an availability event, the latter is + also a reset. + + This operation MAY return as soon as the outage has begun rather than + blocking for the full duration; the TCK does not rely on the response + being delayed. It awaits provider events instead. + + **No shipped scenario reaches this endpoint**, and it is optional for + that reason. The disconnect/reconnect scenario — `PROVIDER_STALE`, then + back to `PROVIDER_READY` — is written as an *unbounded* outage (`the + connection is lost`, then `the connection is restored`), which a TCK + implements with `/stop` followed by `/start`. An earlier version of this + description claimed the TCK used `/restart` for those scenarios; it did + not, and requiring the endpoint on that basis asked every backend author + to implement something nothing called. + + `/start` on reconnect also resets flag state, which `/restart` would + have preserved. That is acceptable today because the only scenario + involved asserts events and client state, never a resolved value. It + stops being acceptable the moment a scenario asserts what a stale + provider serves *during* an outage — last-known-value caching, which is + held behind the `@caching` capability and not yet written. Such a + scenario needs exactly this endpoint's state preservation, which is why + it remains specified rather than deleted: implement it if you want that + capability testable against your backend later. + parameters: + - name: seconds + in: query + required: false + description: | + How long the backend stays unreachable. Defaults to 5. + + Providers differ enormously in how fast they notice an outage — + a streaming provider may see it in milliseconds while a polling + provider needs up to a full poll interval. Feature files therefore + parameterise this value and provider authors tune the matching + await timeouts. + schema: + type: integer + format: int32 + minimum: 0 + default: 5 + example: 5 + responses: + "200": + description: Outage started (and, for blocking implementations, ended). + + /change: + post: + tags: [flags] + operationId: change + summary: "[REQUIRED] Mutate flag configuration so the provider observes a change" + description: | + Mutates the flag configuration such that a conforming provider observes a + configuration change and, on re-evaluation, resolves a **different value** + for the affected flag. + + The implementation MUST: + + * change the resolved value of the flag with key `changing-flag`; + * do so without restarting the backend process, so that a provider sees + a configuration-change signal rather than a reconnect; + * make the change durable until the next `/start` or `/reset`; + * **not return until the new value is actually being served.** + + That last point is the same promise `/start` makes, and it means the + backend, not the provider. A fresh evaluation against the backend MUST + resolve the new value once this returns; how long the *provider under + test* takes to notice is a property of its transport — streaming sees it + in milliseconds, a poller may need most of an interval — and that is what + the TCK's event timeout is for. The two must not be confused: a backend + that returns before it is serving the new value makes the provider's + detection latency unmeasurable, because the clock starts before there is + anything to detect. + + The implementation SHOULD toggle between exactly two known values so that + repeated calls are meaningful and the test remains deterministic + regardless of how many times it has run against the same stack. The + reference implementation toggles `changing-flag`'s `defaultVariant` + between `foo` and `bar`. + + The TCK uses this to assert `PROVIDER_CONFIGURATION_CHANGED`, that the + changed flag key appears in the event payload, and that a subsequent + evaluation returns the new value. + responses: + "200": + description: Flag configuration mutated. + + /reset: + post: + tags: [flags] + operationId: reset + summary: "[OPTIONAL] Restore the seeded baseline without an outage" + description: | + Restores flag state to the baseline of the configuration currently in + effect, discarding any mutation applied by `/change`, **without** making + the backend unreachable at any point. + + This is the preferred scenario-isolation primitive: unlike `/start` it + causes no availability blip, so it cannot inject spurious lifecycle + events into the next scenario. + + **MUST NOT return until the restored baseline is actually being served**, + exactly as `/start` must not. This is worth stating separately because + `/reset` is the endpoint a TCK calls before *every* scenario: a window + here is re-rolled per scenario rather than once per suite, so a small + probability becomes a near-certainty over a few hundred scenarios, and + the failures land on scenarios chosen at random. Two runs of the same + suite then disagree about which scenarios failed, which reads as a flaky + provider and is the hardest shape of this defect to diagnose. + + **Scope.** This operation resets flag state only. It MUST NOT be + expected to start a backend that is currently stopped — that is what + `/start` is for. A TCK therefore uses `/reset` only when the backend is + known to be running, and `/start` otherwise. The reference client tracks + this: `/stop` and `/restart` mark the backend as possibly-unreachable, so + the scenario that follows either of them is prepared with `/start`. + + **Fallback when not implemented.** A backend that does not implement this + operation MUST respond `404` or `501`. The TCK then falls back to + `POST /start?config={defaultConfig}`, which resets flag state at the cost + of a process restart. The fallback is detected once per suite and cached. + + Implementing `/reset` is RECOMMENDED for providers whose reconnect + behaviour makes the `/start` blip hard to distinguish from a real event. + responses: + "200": + description: Flag state restored to the baseline. + "404": + description: Not implemented; the TCK falls back to `/start`. + "501": + description: Not implemented; the TCK falls back to `/start`. + + /healthz: + get: + tags: [health] + operationId: health + summary: "[OPTIONAL] Readiness of the control API" + description: | + Reports whether the control API is ready to accept commands. + + **Fallback when not implemented.** Readiness defaults to "the control + port accepts a TCP connection", which the TCK establishes with a + Testcontainers listening-port wait strategy before the first scenario. A + `404` here is therefore not a failure, and the reference implementation + does not serve this path. + + Note this reports the health of the **control API**, not of the backend. + The backend is deliberately unhealthy during outage scenarios while the + control API must stay reachable — otherwise the TCK could not end the + outage. + responses: + "200": + description: Control API ready. + content: + application/json: + schema: + $ref: "#/components/schemas/Health" + "404": + description: Not implemented; readiness falls back to a TCP port check. + "503": + description: Control API not ready yet. + +components: + schemas: + + Health: + type: object + properties: + status: + type: string + enum: [ok] + description: Present and equal to `ok` when the control API is ready. + required: [status] + + Error: + type: object + properties: + message: + type: string + description: Human-readable explanation. Never interpreted by the TCK. + required: [message] diff --git a/version.txt b/version.txt new file mode 100644 index 00000000..ac39a106 --- /dev/null +++ b/version.txt @@ -0,0 +1 @@ +0.9.0