From 57dae4362498f7f9229a2f11bbd5ce2de6cf852f Mon Sep 17 00:00:00 2001 From: Eric Moore Date: Fri, 25 Sep 2026 17:58:10 -0500 Subject: [PATCH 01/19] feat(testing): CSD flows run on the five-platform matrix MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A CSD reaches `testable` when its flow runs on the matrix (CSD.md §1). Nothing ran one here: flow_spec.py was vendored, but no leg drove a flow. - testing/flows/: the home for flows. Each names its CSD (`csd: CSD-005`); the runner reads that CSD's `shows:` (field ids for `relation`) and `states:` (tags for `state:`). A missing or unparseable CSD, or a flow naming a `proposed:` tag, is a load error before anything boots. - flow_spec.py local delta (VENDORED.md): the `csd:` key, CSD binding at load, and `state:` actually checked. Upstream's body was `pass`. - run_flows.py: pass / fail / refused (by `client:` floor) / cannot-start. Refused stays green; cannot-start is red, since with the floor met it means the flow silently never ran. - run_platform --flows: after a clean smoke walk, in the same app, one sign-in (session_fixture). Flows load before bring-up. All five legs in five-platform-live-qa.yml pass it, and PyYAML is installed per job. - Seeded flow: people.yaml (CSD-005, client >=0.5.224). On a bare node it checks landing with the add card, a search with no match (state: empty), and a cleared search. - testing/test_flows.py (35 tests), wired into build.yml. Every rule was mutation-checked red. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01XXpYDXz2XUwUCG1ePtmVMK --- .github/workflows/build.yml | 3 +- .github/workflows/five-platform-live-qa.yml | 37 +- testing/flows/README.md | 138 +++++++ testing/flows/people.yaml | 62 +++ testing/gate/VENDORED.md | 18 + testing/gate/csd_doc.py | 151 ++++++++ testing/gate/flow_spec.py | 118 +++++- testing/gate/run_flows.py | 244 ++++++++++++ testing/gate/run_platform.py | 53 +++ testing/gate/session_fixture.py | 6 +- testing/test_flows.py | 409 ++++++++++++++++++++ 11 files changed, 1227 insertions(+), 12 deletions(-) create mode 100644 testing/flows/README.md create mode 100644 testing/flows/people.yaml create mode 100644 testing/gate/csd_doc.py create mode 100644 testing/gate/run_flows.py create mode 100644 testing/test_flows.py diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index a28502c0..28adb51b 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -110,7 +110,8 @@ jobs: run: | python3 -m pip install --quiet pytest pyyaml python3 -m pytest testing/test_driver_rules.py testing/test_gate_vendoring.py \ - testing/test_bringup.py testing/test_five_platform_workflow.py -q + testing/test_bringup.py testing/test_five_platform_workflow.py \ + testing/test_flows.py -q - name: A published version must offer a universal wheel # Only for versions already on the index — a PR's VERSION is normally diff --git a/.github/workflows/five-platform-live-qa.yml b/.github/workflows/five-platform-live-qa.yml index bb85b291..fb4f218c 100644 --- a/.github/workflows/five-platform-live-qa.yml +++ b/.github/workflows/five-platform-live-qa.yml @@ -141,6 +141,12 @@ jobs: with: python-version: '3.10' + # THE CSD FLOWS NEED PyYAML (testing/gate/flow_spec.py reads the flow and + # its CSD's typed blocks). Everything else the legs run is stdlib; this is + # the one install, and it goes into the interpreter every leg below uses. + - name: The flow runner's one dependency + run: python3 -m pip install --quiet pyyaml + - name: The node this client will be a client of id: node env: @@ -224,12 +230,18 @@ jobs: - name: Start the node uses: ./.github/actions/ciris-node + # THE SMOKE WALK, THEN THE CSD FLOWS, IN ONE PROCESS AGAINST ONE APP. + # `--flows` runs every testing/flows/*.yaml after the walk passes, in the + # app the walk just proved, signing in once (session_fixture). A flow + # refused by its `client:` floor is reported and does not redden the leg; + # a flow that fails, or never reaches its first screen, does. See + # testing/flows/README.md. Every leg below passes the same flag. - name: Linux desktop run: | python3 -m testing.gate.run_platform --platform desktop --xvfb \ --jar "${{ steps.art.outputs.jar }}" \ --node-version "${{ steps.node.outputs.version }}" \ - --shots shots --report reports/linux.json + --shots shots --report reports/linux.json --flows testing/flows # WITHOUT THIS THE EMULATOR RUNS IN SOFTWARE AND DIES. # @@ -289,7 +301,7 @@ jobs: # # after the emulator had booted and the whole leg had been paid for. # It reads worse on one line and it is the form that runs. - script: python3 -m testing.gate.run_platform --platform android --apk "${{ steps.art.outputs.apk }}" --node-version "${{ steps.node.outputs.version }}" --shots shots --report reports/android.json; rc=$?; adb logcat -d > logcat.txt 2>&1; exit $rc + script: python3 -m testing.gate.run_platform --platform android --apk "${{ steps.art.outputs.apk }}" --node-version "${{ steps.node.outputs.version }}" --shots shots --report reports/android.json --flows testing/flows; rc=$?; adb logcat -d > logcat.txt 2>&1; exit $rc # ALWAYS. A failure you cannot diagnose from the artifact costs a re-run # to learn what this run already knew. @@ -325,6 +337,17 @@ jobs: distribution: temurin java-version: '17' + # A PROVISIONED INTERPRETER for the macOS desktop leg, so the flow runner's + # PyYAML goes into a Python this job owns rather than the image's + # externally-managed one. The iOS leg switches to 3.10 further down and + # installs it again there. + - uses: actions/setup-python@v5 + with: + python-version: '3.12' + + - name: The flow runner's one dependency + run: python3 -m pip install --quiet pyyaml + - name: The node this client will be a client of id: node env: @@ -350,7 +373,7 @@ jobs: python3 -m testing.gate.run_platform --platform desktop \ --jar "$(python3 -m testing.gate.candidate_artifacts --kind desktop | tail -1)" \ --node-version "${{ steps.node.outputs.version }}" \ - --shots shots --report reports/macos.json + --shots shots --report reports/macos.json --flows testing/flows # ── THE iOS BUNDLE, MATERIALIZED THE WAY THE AGENT'S GATE DOES IT ────── # @@ -541,6 +564,9 @@ jobs: - name: iOS simulator run: | set -euo pipefail + # `python3` is the 3.10 set up above now, not the 3.12 the macOS leg + # installed PyYAML into — the flow runner needs it here too. + python3 -m pip install --quiet pyyaml # THE TASK NAME, AND THE DIRECTORY, BOTH WRONG — AND THE SECOND HID THE FIRST. # # `:shared:assembleDebugXCFramework` does not exist. The framework is @@ -632,7 +658,7 @@ jobs: rc=0 python3 -m testing.gate.run_platform --platform ios --app "$app" --udid "$UDID" \ --node-version "${{ steps.node.outputs.version }}" \ - --shots shots --report reports/ios.json || rc=$? + --shots shots --report reports/ios.json --flows testing/flows || rc=$? # THE APP'S OWN ACCOUNT, EITHER WAY. Run 35359571538 got the iOS app # to a real screen — "Engine Failed to Start: server did not become @@ -732,10 +758,11 @@ jobs: - name: Windows desktop shell: bash run: | + python3 -m pip install --quiet pyyaml python3 -m testing.gate.run_platform --platform desktop \ --jar "$(python3 -m testing.gate.candidate_artifacts --kind desktop | tail -1)" \ --node-version "${{ steps.node.outputs.version }}" \ - --shots shots --report reports/windows.json + --shots shots --report reports/windows.json --flows testing/flows - if: always() uses: actions/upload-artifact@v4 diff --git a/testing/flows/README.md b/testing/flows/README.md new file mode 100644 index 00000000..a8807ff2 --- /dev/null +++ b/testing/flows/README.md @@ -0,0 +1,138 @@ +# CSD flows + +A CSD's §4 says what its surface must do. A flow here is that §4 made +executable, and the five-platform gate (`five-platform-live-qa.yml`) runs every +flow in this directory on every leg — Linux, macOS and Windows desktops, the +Android emulator, the iOS simulator — against a real `ciris-server`. + +That is what `testable` means in `CSD.md` §1: *"floor flips off unreleased; flow +runs on the matrix"*. + +## The file + +```yaml +flow: people # the flow's id; unique across this directory +csd: CSD-005 # the CSD it tests — REQUIRED here +title: People on a node with no contacts yet +description: >- # optional; say what is NOT driven and why + … +client: ">=0.5.224" # the client that carries every tag it names + +steps: + - step_id: landing + title: Signing in on a bare node lands on People + requires: # checked BEFORE the step; the first step's is the entry + screen: Contacts + do: # click / input / scroll_to / wait (one per entry) + - wait: card_contacts_add + wait_ms: 5000 # a `wait` waits wait_ms × 4 + expect: # checked AFTER + visible: [card_contacts_add] + absent: [contacts_empty] + + - step_id: search_no_match + title: A search nothing matches is the empty state + do: + - input: {input_contacts_search: "zz-no-such-contact"} + expect: + state: empty # read from the CSD's `states:` block +``` + +The language is `testing/gate/flow_spec.py`'s — vendored from CIRISAgent, with +the CSD/3 §3 predicates (`count`, `number`, `matches`, `one_of`, `each`, +`relation`, `state`) and one local addition, `csd:` (see +`testing/gate/VENDORED.md`). Unknown keys anywhere are a load error. + +### How a flow is tied to its CSD + +`csd: CSD-NNN` resolves to the one `FSD/CSD/CSD-NNN-*.md`, and the runner reads +two of its typed blocks (`testing/gate/csd_doc.py`): + +- **`csd:shows`** → field id → tag. A `relation:` names `ceg:` field ids + (`capacity:composite`), not tags, and this is how they resolve. +- **`csd:states`** → state → tag. `state: empty` holds when the `empty` tag is on + screen **and no other state's tag is** — so an error cannot pass for "nothing + here". + +Each of these is a **load error**, found before any app is started: + +- the CSD does not exist, is ambiguous, or a typed block does not parse; +- the flow names a tag the CSD still marks `proposed:` — in `requires`, + `expect` or `do`. No client carries it, and it would fail as "element not + found", which looks exactly like a broken app; +- `state: X` where the CSD gives X no tag, or a proposed one; +- a `relation:` operand that is not one of the CSD's `shows:` fields. + +`testing/test_flows.py` also checks that every literal tag a flow here names is +a string in the client's `commonMain` source. + +## Verdicts + +| verdict | when | leg | +|---|---|---| +| `pass` | every step held | green | +| `refused` | the `client:` floor is above the client under test (`>=X`, `>X`, `unreleased`) | green — reported, not passed | +| `cannot-start` | floor met, but the first step's `requires` never held | **red** | +| `fail` | it started and a step broke | **red** | + +`cannot-start` is red on purpose. The floor is how a flow waits for a surface +that has not shipped; once the floor is met, a flow that never reached its first +screen is a flow that silently never ran. + +Each leg's report (`reports/.json`) carries a `flows` list with every +outcome and its step-level detail; screenshots and per-flow JSON land under +`shots/flows-/`. The client version the floor is checked against is this +tree's `VERSION` — on the matrix the artifact is asserted to be this tree's. + +## From `building` to `testable` + +1. Every tag the flow needs is real: no `proposed:` left on the rows it drives, + and the PR that adds them has shipped. +2. Write `testing/flows/.yaml` with `csd:` pointing at the CSD and + `client: ">="`. Until a release carries it, use + `client: unreleased` — the flow loads, is checked, and is refused on the + matrix instead of failing. +3. Run it locally (below), then let the nightly matrix run it. +4. When it is **green on the platforms the CSD's §5 declares**, the pen-holder + moves `stage:` to `testable`. Nothing advances the stage automatically — a + green run is evidence for the edit, not the edit. + +## Running one flow locally, against a desktop client + +The runner drives whatever client answers on the test-automation port. Keep it +off your own install: the node takes `--home`, the client reads `CIRIS_HOME` +(`testing/gate/session_fixture.py` explains why they differ). + +```bash +# 1. a throwaway node +python3 -m testing.gate.node_fixture --version v0.5.224 --home /tmp/flows-node \ + --platform x86_64-unknown-linux-gnu # or aarch64-apple-darwin + +# 2. the desktop client in test mode, with its own home +( cd client && ./gradlew :desktopApp:packageUberJarForCurrentOS ) +CIRIS_TEST_MODE=true CIRIS_HOME=/tmp/flows-client \ + java -jar "$(python3 -m testing.gate.candidate_artifacts --kind desktop | tail -1)" & + +# 3. the flow — it signs in (running first-run setup if the node has no owner) +python3 -m testing.gate.run_flows --platform desktop \ + --flows testing/flows/people.yaml --report /tmp/flows.json +``` + +`--client-version` checks floors against another version (default: `VERSION`); +`--no-sign-in` drives whatever screen the client is already on; `--url` points at +a test server other than `http://127.0.0.1:9091`. Exit status is 0 only when +every flow passed or was refused. + +On the matrix the same code runs inside `run_platform` (`--flows testing/flows`) +after the smoke walk, in the app the walk just brought up, so there is one +bring-up per leg and one session. + +## What this does not do yet + +- **No navigation.** A flow's first step names its screen and the runner waits + for it; it does not walk the sidebar there. Every flow today starts where + sign-in lands (`Contacts`). A flow for another surface needs `nav_map` to drive + the hop — the CSD names the surface, and the hop is derived, never written. +- **Only what a bare node can show.** The matrix stands up one node with no + contacts, no agent and no peers, so CSD-005's populated list and receipt sheet + are not driven here. diff --git a/testing/flows/people.yaml b/testing/flows/people.yaml new file mode 100644 index 00000000..7fb599f4 --- /dev/null +++ b/testing/flows/people.yaml @@ -0,0 +1,62 @@ +flow: people +csd: CSD-005 +title: People on a node with no contacts yet +description: >- + The matrix's first CSD flow, and the one that proves the wiring: every leg signs + in on a bare ciris-server, which has no contacts, so People lands in the + "empty and unsearched" shape CSD-005 §4 describes — the add card INSTEAD of an + empty block. A search nothing matches then shows the empty state proper, and + clearing it brings the add card back. Every tag here ships in 0.5.224 + (PeopleTags in ContactsScreen/PeopleSupport.kt). + + Not driven here, and why: the populated list and the receipt sheet need a + second node to be a contact of, which the matrix does not stand up; the + `proposed:` trust chip cannot be named by a flow at all. +client: ">=0.5.224" + +steps: + - step_id: landing + title: Signing in on a bare node lands on People, offering to add someone + description: >- + `requires` is the entry precondition, stated rather than assumed: a client + that landed anywhere else reports "cannot start", not "a People element is + broken". The wait absorbs the first /v1/contacts read (the loading state). + requires: + screen: Contacts + do: + - wait: card_contacts_add + wait_ms: 5000 + expect: + screen: Contacts + visible: [input_contacts_search, card_contacts_add, input_contacts_add_key, + btn_contacts_add_submit, btn_contacts_refresh] + # The add card stands in for the empty block over an EMPTY, UNSEARCHED + # list; and a node that answered is neither loading, nor in error, nor + # too old for contacts. + absent: [contacts_empty, contacts_list, contacts_loading, contacts_error, + contacts_unsupported] + + - step_id: search_no_match + title: A search nothing matches is the empty state, not an error + description: >- + `state: empty` is checked against CSD-005's `states:` block: contacts_empty + on screen, and the populated, loading and error tags all absent — so an + error cannot pass for "no matches". + do: + - input: {input_contacts_search: "zz-no-such-contact"} + - wait: contacts_empty + wait_ms: 2500 + expect: + state: empty + absent: [card_contacts_add] + + - step_id: clear_search + title: Clearing the search brings the add card back + do: + - input: {input_contacts_search: ""} + - wait: card_contacts_add + wait_ms: 2500 + expect: + screen: Contacts + visible: [card_contacts_add] + absent: [contacts_empty, contacts_error] diff --git a/testing/gate/VENDORED.md b/testing/gate/VENDORED.md index 122eb552..a4b58050 100644 --- a/testing/gate/VENDORED.md +++ b/testing/gate/VENDORED.md @@ -67,6 +67,24 @@ tagged is drivable" about a screen with no elements on it. `check_csd.py`, and two implementations of one DSL is the drift this repo exists to measure. +### `flow_spec.py` — local delta: a flow names its CSD (`csd:`) + +Added so CSD flows can run on this repo's matrix (`testing/flows/`, +`testing/gate/run_flows.py`). Each change is marked `LOCAL DELTA` in the source: + +| change | reason | +|---|---| +| `csd` added to `_FLOW_KEYS`; `FlowSpec.csd_id` / `FlowSpec.csd` | a flow says which CSD it tests, so the runner can read that CSD's `shows:` (→ `field_tags`, for `relation`) and `states:` (→ `state_tags`, for `state:`) instead of every caller wiring them by hand | +| `FlowSpec.load(path, csd_root=None)` binds the CSD at load | a CSD that is missing, ambiguous or does not parse is a **load error**, never a skip — a flow that silently loses its CSD loses the checks that make `state:`/`relation:` mean anything. The CSD is read by `csd_doc.py` (ours), which takes its block grammar from `packaging/check_csd_v3.py` rather than re-typing it | +| a `proposed:` tag named in `requires`/`expect`/`do`, a `state:` whose tag is proposed or absent, or a `relation` operand that is not a `shows:` field → load error | no client carries a proposed tag, so asserting it fails as "element not found" — indistinguishable from a broken app. `count`/`each` globs are deliberately NOT checked: CSD-005's real `contacts_row_*` rows share a prefix with its proposed `contacts_row_trust` chip | +| `state:` is now CHECKED: that state's tag on screen, every other state's tag not; with no `states:` map it fails | upstream's `state:` body was `pass` — the one predicate CSD/3 makes mandatory asserted nothing, a vacuous green. Upstream flows do not use `state:`, so none of them changes behaviour | +| `FlowRunner(state_tags=…)`; `run()` fills both maps from `spec.csd` when the caller did not | one source for the maps: the CSD | +| `write_report` records `csd` | a result that cannot say what it tested cannot be acted on | + +A flow without `csd:` still loads and runs exactly as before; `run_flows.py` +is what requires the key for flows in this repo. Upstream needs the same key +before CIRISAgent's flows can be checked against their CSDs the same way. + ### `platforms.py` — and a mistake this file previously recorded as a fact An earlier version of this document said all three vendored modules were diff --git a/testing/gate/csd_doc.py b/testing/gate/csd_doc.py new file mode 100644 index 00000000..0ec74cde --- /dev/null +++ b/testing/gate/csd_doc.py @@ -0,0 +1,151 @@ +"""The half of a CSD a flow needs: its `shows:` fields and its `states:` tags. + +A flow in `testing/flows/` names its CSD (`csd: CSD-005`). This module finds that +document and reads the two typed blocks the runner cannot work without: + + * `csd:shows` -> `field_tags`, the `ceg:` field id -> the tag drawing it. A + `relation:` operand is a field id, so without this a flow can only relate + boxes rather than constitutional values (CSD/3 §3). + * `csd:states` -> `state_tags`, state -> tag. `state: empty` is asserted as + "that state's tag is on screen, and every other state's tag is not" — so an + error cannot pass for an empty list, which is the one confusion CSD/3 §2.2 + makes mandatory to prevent. + +It also records which tags are still `proposed:`, because a flow may not drive or +assert a tag no client has shipped: that would fail as "element not found", which +is indistinguishable from a broken app. + +ONE BLOCK GRAMMAR. The fenced-block pattern is `packaging/check_csd_v3.py`'s own +`BLOCK`, loaded from that file rather than re-typed here, so the checker and the +runner cannot disagree about what counts as a typed block. (It is loaded by path: +`import packaging` would find the PyPI package of that name first.) + +EVERY FAILURE IS A LOAD ERROR. A CSD that is missing, ambiguous or does not parse +raises `CsdError` — never a silent skip, because a flow that quietly loses its CSD +loses the checks that make its `state:` and `relation:` mean anything. +""" + +from __future__ import annotations + +import importlib.util +import re +from dataclasses import dataclass, field +from pathlib import Path +from typing import Dict, Optional, Set + +import yaml + +#: Where this repo keeps its CSDs. +DEFAULT_CSD_ROOT = Path(__file__).resolve().parents[2] / "FSD" / "CSD" +_CHECKER = Path(__file__).resolve().parents[2] / "packaging" / "check_csd_v3.py" +_ID = re.compile(r"^CSD-\d{3}$") +PROPOSED = "proposed:" + + +class CsdError(Exception): + """The CSD a flow names cannot be used. Always a load error.""" + + +def _block_pattern() -> "re.Pattern[str]": + spec = importlib.util.spec_from_file_location("_check_csd_v3", _CHECKER) + if spec is None or spec.loader is None: # pragma: no cover — the file is in the tree + raise CsdError(f"cannot load the CSD block grammar from {_CHECKER}") + mod = importlib.util.module_from_spec(spec) + spec.loader.exec_module(mod) + return mod.BLOCK + + +BLOCK = _block_pattern() + + +@dataclass +class CsdDoc: + csd_id: str + path: Path + stage: Optional[str] + #: `ceg:` field id -> tag. A parameterised id is keyed both as written + #: (`consent:{kind}`) and as bound (`consent:replication`). + field_tags: Dict[str, str] = field(default_factory=dict) + #: state name -> tag, for the states that name one. + state_tags: Dict[str, str] = field(default_factory=dict) + #: Every tag the CSD marks `proposed:` (prefix stripped). + proposed: Set[str] = field(default_factory=set) + #: Field ids whose tag is still proposed. + proposed_fields: Set[str] = field(default_factory=set) + #: State names whose tag is still proposed. + proposed_states: Set[str] = field(default_factory=set) + + +def _bound(ceg: str, bind: Dict[str, str]) -> str: + out = ceg + for k, v in bind.items(): + out = out.replace("{" + str(k) + "}", str(v)) + return out + + +def parse(path: Path, csd_id: str = "") -> CsdDoc: + """Read one CSD's typed blocks. Raises CsdError on anything unusable.""" + try: + text = path.read_text(encoding="utf-8") + except OSError as e: + raise CsdError(f"{path}: cannot be read ({e})") from e + + blocks: Dict[str, object] = {} + for name, body in BLOCK.findall(text): + try: + blocks[name] = yaml.safe_load(body) + except yaml.YAMLError as e: + raise CsdError(f"{path}: its `csd:{name}` block does not parse: {e}") from e + if "stage" not in blocks: + # Every CSD/3 document has one; a file without it is not a CSD, and a + # flow bound to it would be bound to prose. + raise CsdError(f"{path}: no `yaml csd:stage` block — not a CSD/3 document") + + doc = CsdDoc(csd_id=csd_id or path.stem, path=path, + stage=(blocks.get("stage") or {}).get("stage")) + + shows = blocks.get("shows") or {} + if not isinstance(shows, dict): + raise CsdError(f"{path}: `csd:shows` is not a mapping") + for i, f in enumerate(shows.get("fields") or []): + if not isinstance(f, dict) or not f.get("ceg"): + raise CsdError(f"{path}: csd:shows field[{i}] has no `ceg:`") + raw_tag = str(f.get("tag") or "") + if not raw_tag: + continue + proposed = raw_tag.startswith(PROPOSED) + tag = raw_tag[len(PROPOSED):] if proposed else raw_tag + ids = {str(f["ceg"]), _bound(str(f["ceg"]), f.get("bind") or {})} + for fid in ids: + doc.field_tags[fid] = tag + if proposed: + doc.proposed_fields.add(fid) + if proposed: + doc.proposed.add(tag) + + states = blocks.get("states") or {} + if not isinstance(states, dict): + raise CsdError(f"{path}: `csd:states` is not a mapping") + for name, row in states.items(): + raw_tag = str((row or {}).get("tag") or "") if isinstance(row, dict) else "" + if not raw_tag: + continue + if raw_tag.startswith(PROPOSED): + doc.proposed.add(raw_tag[len(PROPOSED):]) + doc.proposed_states.add(str(name)) + continue # a proposed state tag cannot be asserted; see flow_spec + doc.state_tags[str(name)] = raw_tag + return doc + + +def load(csd_id: str, root: Optional[Path] = None) -> CsdDoc: + """Find `CSD-NNN-*.md` under `root` and parse it.""" + if not isinstance(csd_id, str) or not _ID.match(csd_id): + raise CsdError(f"`csd: {csd_id!r}` is not a CSD id; write it as e.g. `CSD-005`") + root = Path(root) if root else DEFAULT_CSD_ROOT + hits = sorted(root.glob(f"{csd_id}-*.md")) + if not hits: + raise CsdError(f"{csd_id}: no {csd_id}-*.md under {root} — a flow cannot name a CSD that does not exist") + if len(hits) > 1: + raise CsdError(f"{csd_id}: ambiguous, {len(hits)} documents match: {[h.name for h in hits]}") + return parse(hits[0], csd_id) diff --git a/testing/gate/flow_spec.py b/testing/gate/flow_spec.py index 038c4f89..90e06f77 100644 --- a/testing/gate/flow_spec.py +++ b/testing/gate/flow_spec.py @@ -59,7 +59,9 @@ "count", "number", "matches", "one_of", "each", "relation", "state", } _ACTION_KEYS = {"click", "input", "scroll_to", "wait", "wait_ms"} -_FLOW_KEYS = {"flow", "title", "description", "client", "steps"} +#: LOCAL DELTA (VENDORED.md): `csd` names the CSD a flow tests, so the runner can +#: read that CSD's `shows:` (for `relation` field ids) and `states:` (for `state:`). +_FLOW_KEYS = {"flow", "title", "description", "client", "steps", "csd"} _RELATION_OPS = {"eq", "ne", "lt", "lte", "gt", "gte", "min_of", "max_of", "sum_of"} @@ -70,6 +72,14 @@ class SpecError(Exception): """The spec itself is wrong. Raised at load time, before anything is driven.""" +def _tags_of(cond: "Condition") -> List[str]: + """Every literal tag a condition names (globs and field ids excluded).""" + out = list(cond.visible) + list(cond.absent) + for d in (cond.text, cond.number, cond.matches, cond.one_of): + out.extend(d.keys()) + return out + + @dataclass class Condition: """A `requires` or `expect` block: what must be true at a point in the flow.""" @@ -251,9 +261,12 @@ class FlowSpec: description: str = "" client_floor: Optional[str] = None path: Optional[Path] = None + #: LOCAL DELTA: the CSD this flow tests, when it names one (`csd: CSD-005`). + csd_id: Optional[str] = None + csd: Any = None # testing.gate.csd_doc.CsdDoc @classmethod - def load(cls, path: Path) -> "FlowSpec": + def load(cls, path: Path, csd_root: Optional[Path] = None) -> "FlowSpec": try: raw = yaml.safe_load(path.read_text(encoding="utf-8")) except yaml.YAMLError as exc: @@ -274,7 +287,7 @@ def load(cls, path: Path) -> "FlowSpec": if s.step_id in seen: raise SpecError(f"{path}: duplicate step_id {s.step_id!r} (steps {seen[s.step_id]} and {i})") seen[s.step_id] = i - return cls( + spec = cls( flow=str(raw["flow"]), title=str(raw.get("title") or raw["flow"]), description=str(raw.get("description") or ""), @@ -282,6 +295,74 @@ def load(cls, path: Path) -> "FlowSpec": steps=steps, path=path, ) + if "csd" in raw: + spec._bind_csd(raw["csd"], csd_root) + return spec + + def _bind_csd(self, csd_id: Any, csd_root: Optional[Path]) -> None: + """LOCAL DELTA: load the named CSD and hold the flow to it. + + A CSD that is missing or does not parse is a LOAD ERROR, never a skip — + the flow would otherwise run with `state:` and `relation:` unanchored. + And a flow may not name a `proposed:` tag anywhere: no client carries + it, so driving or asserting it fails as "element not found", which is + indistinguishable from a broken app. + """ + from testing.gate.csd_doc import CsdError, load as load_csd # noqa: PLC0415 + + where = f"{self.path}" + try: + doc = load_csd(csd_id, csd_root) + except CsdError as e: + raise SpecError(f"{where}: `csd:` {e}") from e + self.csd_id, self.csd = str(csd_id), doc + + for step in self.steps: + at = f"{where}: step {step.step_id!r}" + for half, cond in (("requires", step.requires), ("expect", step.expect)): + for tag in _tags_of(cond): + if tag in doc.proposed: + raise SpecError( + f"{at}.{half} names {tag!r}, which {doc.csd_id} still marks " + f"`proposed:` — a flow cannot assert a tag no client carries" + ) + # Globs (`count`/`each` `of:`) are NOT checked against the + # proposed set: CSD-005's real `contacts_row_*` rows share a + # prefix with its proposed `contacts_row_trust` chip, and + # refusing the glob would refuse a flow that names no proposed + # tag. `each` over an empty match already fails at run time. + if cond.state is not None: + if cond.state in doc.proposed_states: + raise SpecError( + f"{at}.{half} asserts `state: {cond.state}`, whose tag " + f"{doc.csd_id} still marks `proposed:`" + ) + if cond.state not in doc.state_tags: + raise SpecError( + f"{at}.{half} asserts `state: {cond.state}`, but {doc.csd_id}'s " + f"`states:` names no tag for it, so it cannot be checked" + ) + if cond.relation is not None: + r = cond.relation + operands = [r["left"]] + list(r.get("of") or []) + ( + [r["right"]] if r.get("right") else []) + for fid in map(str, operands): + if fid in doc.proposed_fields: + raise SpecError( + f"{at}.{half} relates {fid!r}, whose tag {doc.csd_id} " + f"still marks `proposed:`" + ) + if ":" in fid and fid not in doc.field_tags: + raise SpecError( + f"{at}.{half} relates {fid!r}, which is not a field " + f"of {doc.csd_id}'s `shows:`" + ) + for action in step.do: + if action.target in doc.proposed: + raise SpecError( + f"{at}.do drives {action.target!r}, which {doc.csd_id} still " + f"marks `proposed:`" + ) def check_client_floor(floor: Optional[str], actual: Optional[str]) -> Optional[str]: @@ -354,7 +435,8 @@ class FlowRunner: """Executes a FlowSpec against a connected DesktopAppHelper.""" def __init__(self, helper, platform=None, artifacts: Optional[Path] = None, - field_tags: Optional[Dict[str, str]] = None) -> None: + field_tags: Optional[Dict[str, str]] = None, + state_tags: Optional[Dict[str, str]] = None) -> None: self.helper = helper self.platform = platform self.artifacts = Path(artifacts) if artifacts else None @@ -363,6 +445,9 @@ def __init__(self, helper, platform=None, artifacts: Optional[Path] = None, #: block. `relation` operands are field ids, so without this a flow #: could only relate boxes rather than constitutional values. self.field_tags: Dict[str, str] = dict(field_tags or {}) + #: LOCAL DELTA: state -> tag, from the CSD's `states:` block. Filled + #: from the spec's CSD at `run()` when the caller does not pass it. + self.state_tags: Dict[str, str] = dict(state_tags or {}) async def _drivable(self) -> List[str]: """Tags actually ON SCREEN. The failure message's most useful sentence. @@ -415,7 +500,25 @@ async def _check(self, cond: Condition, label: str) -> Optional[str]: # state's row names. Kept as a plain visibility check rather than a # new mechanism: the CSD already had to name a tag per state, and a # second way to say the same thing is how two spellings drift. - pass # asserted via `visible:`/`absent:` alongside; see CSD/3 §2.3 + # + # LOCAL DELTA. Upstream this was `pass`, so `state:` asserted + # NOTHING — a vacuous green in the one predicate CSD/3 makes + # mandatory. It now reads the CSD's `states:` map: that state's tag + # must be on screen and every OTHER state's tag must not be, so an + # error cannot pass for an empty list. With no map it FAILS rather + # than passing: an unanchored `state:` is unchecked, not true. + want = self.state_tags.get(cond.state) + if not want: + return (f"{label}: `state: {cond.state}` cannot be checked — no CSD " + f"`states:` tag for it (does the flow name its `csd:`?)") + if not await self.helper.is_element_visible(want): + await self.helper.scroll_into_view(want) + if not await self.helper.is_element_visible(want): + return f"{label}: state {cond.state!r} — its tag {want!r} is not on screen" + for other, tag in sorted(self.state_tags.items()): + if other != cond.state and tag != want and await self.helper.is_element_visible(tag): + return (f"{label}: state {cond.state!r} expected, but {other!r}'s " + f"tag {tag!r} is on screen") if cond.count is not None: got = _match_glob(on_screen, cond.count["of"]) @@ -551,6 +654,10 @@ def _shot(self, spec: FlowSpec, step: Step) -> Optional[str]: return str(got) if got else None async def run(self, spec: FlowSpec) -> bool: + if spec.csd is not None: + # LOCAL DELTA: a flow that names its CSD carries its own maps. + self.field_tags = self.field_tags or dict(spec.csd.field_tags) + self.state_tags = self.state_tags or dict(spec.csd.state_tags) print(f"\n FLOW {spec.flow} — {spec.title}") if spec.description: print(f" {spec.description}") @@ -630,6 +737,7 @@ def write_report(self, spec: FlowSpec) -> Optional[Path]: "flow": spec.flow, "title": spec.title, "spec": str(spec.path), + "csd": spec.csd_id, "client_floor": spec.client_floor, "passed": all(r.status != "fail" for r in self.results), "steps": [ diff --git a/testing/gate/run_flows.py b/testing/gate/run_flows.py new file mode 100644 index 00000000..5dea6803 --- /dev/null +++ b/testing/gate/run_flows.py @@ -0,0 +1,244 @@ +"""Run this repo's CSD flows against a live client — the `testable` half of CSD/3. + + # against a client that is already running (a desktop at a keyboard): + python3 -m testing.gate.run_flows --platform desktop --flows testing/flows + + # on the matrix: the same code, called by run_platform after its smoke walk, + # in the same process and against the app that walk just brought up: + python3 -m testing.gate.run_platform --platform desktop --jar … --flows testing/flows + +CSD.md §1: a CSD reaches `testable` when "floor flips off unreleased; flow runs on +the matrix". This is the thing that runs it. + +FOUR VERDICTS PER FLOW, AND ONLY TWO OF THEM ARE GREEN. + + pass every step's requires/do/expect held + refused the flow's `client:` floor is above the client under test. The + flow cannot start HERE, and says why; it is neither passed nor + failed, and it does not redden the leg + cannot-start the floor is met, but the flow's first `requires` never held — + the client never reached the screen the flow starts on + fail it started and a step broke + +`cannot-start` REDDENS THE LEG, deliberately, and differs from CIRISAgent's gate +here. Upstream reports it as a warning because it was their only way to hold a +flow for a surface no release carried. This repo has the `client:` floor for that +(`unreleased`, `>X`). With the floor met, a flow that never reached its first +screen is a flow that silently never ran — and a leg that stays green while its +only flow never ran is the vacuous green this harness exists to refuse. + +LOADING IS ALL-OR-NOTHING, AND IT HAPPENS FIRST. Every flow must parse, name its +CSD, and name no `proposed:` tag before anything is driven: a spec error found +after ten minutes of emulator is ten minutes wasted, and one found after a +partial run hides behind the flows that did run. +""" + +from __future__ import annotations + +import argparse +import asyncio +import json +import sys +import time +from dataclasses import asdict, dataclass, field +from pathlib import Path +from typing import Any, List, Optional, Sequence + +from testing.gate.flow_spec import FlowRunner, FlowSpec, SpecError, check_client_floor, discover + +REPO = Path(__file__).resolve().parents[2] +DEFAULT_FLOWS = REPO / "testing" / "flows" + +PASS, FAIL, REFUSED, CANNOT_START = "pass", "fail", "refused", "cannot-start" +#: The only verdicts that leave a leg green. +GREEN = {PASS, REFUSED} + +#: Screens a signed-out client can be on. Anything else is taken as signed in. +SIGNED_OUT = {"Login", "Setup"} + + +@dataclass +class FlowOutcome: + flow: str + csd: Optional[str] + status: str + detail: str = "" + steps: List[dict] = field(default_factory=list) + report: Optional[str] = None + + +def default_client_version() -> str: + """This tree's version. On the matrix the artifact IS this tree's (asserted + by candidate_artifacts), so the floor is checked against the candidate.""" + try: + return (REPO / "VERSION").read_text(encoding="utf-8").strip() + except OSError: + return "" + + +def load_flows(paths: Sequence[str | Path], csd_root: Optional[Path] = None) -> List[FlowSpec]: + """Every flow, loaded and bound to its CSD — or a SpecError naming the first + that is not. Never a partial list.""" + files = discover([str(p) for p in paths]) + if not files: + raise SpecError(f"no flows found in {[str(p) for p in paths]}") + specs: List[FlowSpec] = [] + seen: dict[str, Path] = {} + for path in files: + spec = FlowSpec.load(path, csd_root=csd_root) + if spec.csd is None: + raise SpecError( + f"{path}: names no `csd:`. Every flow in this repo tests a CSD, and the " + f"runner reads that CSD's `shows:` and `states:` to check it" + ) + if spec.flow in seen: + raise SpecError(f"{path}: flow id {spec.flow!r} is also used by {seen[spec.flow]}") + seen[spec.flow] = path + specs.append(spec) + return specs + + +async def _settle_on(helper, screen: str, timeout: float, poll: float = 1.0) -> str: + """Wait for the flow's starting screen. Not navigation — a landing that is + still composing is not a flow that cannot start. Returns the last screen.""" + deadline = time.monotonic() + timeout + cur = await helper.get_screen() + while cur != screen and time.monotonic() < deadline: + await asyncio.sleep(poll) + cur = await helper.get_screen() + return cur + + +async def run_one(spec: FlowSpec, helper, *, platform=None, artifacts: Optional[Path] = None, + client_version: Optional[str] = None, start_timeout: float = 30.0) -> FlowOutcome: + refusal = check_client_floor(spec.client_floor, client_version) + if refusal: + print(f"\n FLOW {spec.flow} ({spec.csd_id}) — REFUSED by its floor\n {refusal}") + return FlowOutcome(spec.flow, spec.csd_id, REFUSED, refusal) + + start = spec.steps[0].requires.screen + if start: + await _settle_on(helper, start, start_timeout) + + runner = FlowRunner(helper, platform=platform, artifacts=artifacts) + try: + ok = await runner.run(spec) + except Exception as e: # noqa: BLE001 — a crash in a flow is that flow's verdict + ok = False + runner.results.append(_crash(e)) + report = runner.write_report(spec) + steps = [asdict(r) for r in runner.results] + print(f"\n {runner.summary(spec)}") + + if ok: + status, detail = PASS, runner.summary(spec) + else: + bad = next((r for r in reversed(runner.results) if r.status == "fail"), None) + first = runner.results[0] if runner.results else None + if first is not None and len(runner.results) == 1 and first.phase == "requires": + status, detail = CANNOT_START, f"first step {first.step_id!r}: {first.detail}" + else: + status = FAIL + detail = f"step {bad.step_id!r} ({bad.phase}): {bad.detail}" if bad else "failed" + return FlowOutcome(spec.flow, spec.csd_id, status, detail, steps, str(report) if report else None) + + +def _crash(e: Exception): + from testing.gate.flow_spec import StepResult # noqa: PLC0415 + return StepResult("", "the runner raised", "fail", "do", f"{type(e).__name__}: {e}") + + +def leg_ok(outcomes: Sequence[FlowOutcome]) -> bool: + return all(o.status in GREEN for o in outcomes) + + +def summary(outcomes: Sequence[FlowOutcome]) -> str: + counts = {s: sum(1 for o in outcomes if o.status == s) for s in (PASS, FAIL, CANNOT_START, REFUSED)} + return ", ".join(f"{n} {s}" for s, n in counts.items() if n) or "no flows" + + +def sign_in(drv, username: str, password: str) -> str: + """Reuse the session if the client has one; make one if it does not.""" + from testing.gate import session_fixture # noqa: PLC0415 + + screen = drv.screen() + if screen not in SIGNED_OUT: + return screen + session_fixture.run_setup(drv, username, password) + return session_fixture.log_in(drv, username, password) + + +def run_all(specs: Sequence[FlowSpec], drv, *, platform=None, artifacts: Optional[Path] = None, + client_version: Optional[str] = None, username: str = "qaadmin", + password: str = "QaAdmin!2345", establish_session: bool = True, + helper: Any = None) -> List[FlowOutcome]: + """Run every flow in order against one live client. Signs in once, only if + some flow will actually run.""" + from testing.gate.flow_helper import SyncFlowHelper # noqa: PLC0415 + + helper = helper or SyncFlowHelper(drv) + runnable = [s for s in specs if not check_client_floor(s.client_floor, client_version)] + if runnable and establish_session and drv is not None: + landed = sign_in(drv, username, password) + print(f" session: signed in, on {landed!r}") + + async def go() -> List[FlowOutcome]: + return [await run_one(s, helper, platform=platform, artifacts=artifacts, + client_version=client_version) for s in specs] + + outcomes = asyncio.run(go()) + print(f"\n flows: {summary(outcomes)}") + for o in outcomes: + print(f" [{o.status:^12}] {o.flow} ({o.csd}): {o.detail}") + return outcomes + + +def main(argv: Optional[List[str]] = None) -> int: + ap = argparse.ArgumentParser(description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("--platform", default="desktop", choices=("desktop", "android", "ios")) + ap.add_argument("--url", default="http://127.0.0.1:9091", help="the client's test server") + ap.add_argument("--flows", action="append", help=f"a flow or a directory (default {DEFAULT_FLOWS})") + ap.add_argument("--csd-root", type=Path, help="where the CSDs live (default FSD/CSD)") + ap.add_argument("--client-version", default=None, + help="the version under test, for `client:` floors (default: VERSION)") + ap.add_argument("--artifacts", type=Path, help="screenshots and per-flow JSON") + ap.add_argument("--report", type=Path, help="write every outcome here as JSON") + ap.add_argument("--username", default="qaadmin") + ap.add_argument("--password", default="QaAdmin!2345") + ap.add_argument("--no-sign-in", action="store_true", + help="drive whatever screen the client is on; do not make a session") + args = ap.parse_args(argv) + + try: + specs = load_flows(args.flows or [DEFAULT_FLOWS], args.csd_root) + except SpecError as e: + print(f"[FAIL] {e}") + return 1 + print(f"loaded {len(specs)} flow(s): {', '.join(f'{s.flow} ({s.csd_id})' for s in specs)}") + + from testing.driver import DriverError, TestAutomationServer # noqa: PLC0415 + from testing.gate.platforms import build_platform # noqa: PLC0415 + from testing.gate.session_fixture import SessionUnavailable # noqa: PLC0415 + + drv = TestAutomationServer(base_url=args.url) + version = args.client_version if args.client_version is not None else default_client_version() + artifacts = args.artifacts or Path("shots") / f"flows-{args.platform}" + try: + drv.wait_for_server(timeout=30) + outcomes = run_all(specs, drv, platform=build_platform(args), artifacts=artifacts, + client_version=version, username=args.username, + password=args.password, establish_session=not args.no_sign_in) + except (DriverError, SessionUnavailable) as e: + print(f"[FAIL] {e}") + return 1 + if args.report: + args.report.parent.mkdir(parents=True, exist_ok=True) + args.report.write_text(json.dumps([asdict(o) for o in outcomes], indent=2), encoding="utf-8") + ok = leg_ok(outcomes) + print(f"flows: {'PASS' if ok else 'FAIL'}") + return 0 if ok else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/testing/gate/run_platform.py b/testing/gate/run_platform.py index 1f0dc384..b78db601 100644 --- a/testing/gate/run_platform.py +++ b/testing/gate/run_platform.py @@ -62,6 +62,8 @@ class Report: artifact: str = "" node_version: str = "" steps: list[StepResult] = field(default_factory=list) + #: One entry per CSD flow (run_flows.FlowOutcome), when `--flows` was given. + flows: list[dict] = field(default_factory=list) def add(self, name: str, ok: bool, detail: str = "", shot: str | None = None) -> None: self.steps.append(StepResult(name, ok, detail, shot)) @@ -247,6 +249,11 @@ def main() -> int: ap.add_argument("--report", type=Path) ap.add_argument("--node-version", default="") ap.add_argument("--timeout", type=float, default=120.0) + ap.add_argument("--flows", action="append", + help="after the smoke walk, run these CSD flows (a file or a directory; " + "repeatable) against the same app — see testing/flows/README.md") + ap.add_argument("--client-version", default=None, + help="the version under test, for flows' `client:` floors (default: VERSION)") args = ap.parse_args() rep = Report(platform=args.platform, node_version=args.node_version) @@ -255,6 +262,18 @@ def main() -> int: from testing.gate.platforms import build_platform platform = build_platform(args) + # FLOWS LOAD BEFORE ANYTHING BOOTS. A flow that does not parse, or names a + # CSD that does not, is found in a second — not after the emulator. + specs = None + if args.flows: + from testing.gate import run_flows + from testing.gate.flow_spec import SpecError + try: + specs = run_flows.load_flows(args.flows) + except SpecError as e: + rep.add("flows-load", False, str(e)) + return _finish(args, rep) + plan = None try: plan = plan_for(args) @@ -266,6 +285,8 @@ def main() -> int: # PROVEN, NOT ASSUMED. drv.wait_for_server(timeout=args.timeout) walk(drv, rep, args.shots, platform, args_timeout=args.timeout) + if specs is not None: + flows(drv, rep, specs, args, platform) rep.ok = all(s.ok for s in rep.steps) except bringup.CannotRun as e: # LOUD. Not a skip: the caller decides what to exclude, and it does so @@ -288,7 +309,39 @@ def main() -> int: # check=False: teardown runs after failures too, and one that fails # must not hide the failure that caused it. bringup.run(td, check=False) + return _finish(args, rep) + +def flows(drv: TestAutomationServer, rep: Report, specs, args, platform) -> None: + """The CSD flows, in the SAME app the walk just proved — no second bring-up. + + Only after a clean walk: the walk is what proves there is a composed app on + a real node to drive, and flows run on anything less would fail for the + walk's reason under their own names. + """ + from testing.gate import run_flows + from testing.gate.session_fixture import SessionUnavailable + + if not all(s.ok for s in rep.steps): + rep.add("flows", False, "not run — the smoke walk above failed, so there is no app to drive") + return + # Per LEG, not per platform: the linux desktop and the android emulator + # share a runner and a `shots/` directory, and must not overwrite each other. + leg = args.report.stem if args.report else args.platform + version = args.client_version if args.client_version is not None else run_flows.default_client_version() + try: + outcomes = run_flows.run_all(specs, drv, platform=platform, + artifacts=args.shots / f"flows-{leg}", + client_version=version) + except SessionUnavailable as e: + rep.add("flows", False, f"no session to run them in: {e}") + return + rep.flows = [asdict(o) for o in outcomes] + rep.add("flows", run_flows.leg_ok(outcomes), run_flows.summary(outcomes)) + + +def _finish(args, rep: Report) -> int: + rep.ok = rep.ok and all(s.ok for s in rep.steps) if args.report: # MAKE THE DIRECTORY. `--report reports/.json` names a path in a # directory nothing creates: the workflow passes it, the artifact upload diff --git a/testing/gate/session_fixture.py b/testing/gate/session_fixture.py index 62e2aa20..8f52d205 100644 --- a/testing/gate/session_fixture.py +++ b/testing/gate/session_fixture.py @@ -79,7 +79,11 @@ def run_setup(drv: TestAutomationServer, username: str, password: str, if "txt_owner_hint" in _tags(drv): return # already owned; nothing to do - drv.click("btn_local_login") + # Desktop's first run shows Login; a client whose first run opens the wizard + # directly is already where this click would take it, and clicking a + # `btn_local_login` that is not on screen fails the fixture for nothing. + if drv.screen() != "Setup": + drv.click("btn_local_login") if not _settle(drv, "Setup", timeout=30): raise SessionUnavailable( f"btn_local_login did not reach Setup (on {drv.screen()!r}); on a node " diff --git a/testing/test_flows.py b/testing/test_flows.py new file mode 100644 index 00000000..a324f4b0 --- /dev/null +++ b/testing/test_flows.py @@ -0,0 +1,409 @@ +"""CSD flows on the matrix: loading, the floor, and the runner's verdicts. + +Every rule here has a red path, and each test is the red path — a check whose +failing half has never run is a check with an untested half (AGENTS.md). The +runner tests drive `run_flows.run_one` through a FAKE helper, which is honest +here in a way it is not for the driver: what is under test is the verdict logic +over `/tree`-shaped answers, not the transport (test_driver_rules.py owns that). +""" + +from __future__ import annotations + +import asyncio +import re +import textwrap +from dataclasses import dataclass +from pathlib import Path +from typing import Dict, Optional + +import pytest +import yaml + +from testing.gate import run_flows +from testing.gate.flow_spec import FlowSpec, SpecError + +REPO = Path(__file__).resolve().parents[1] +FLOWS = REPO / "testing" / "flows" + +CSD_TEXT = """\ +# CSD-900 — a test surface + +```yaml csd:stage +stage: sketched +owner: CIRISClient +``` + +```yaml csd:shows +fields: + - ceg: x_private:score + use: display-only + type: float + example: 0.5 + renders: "0.5" + tag: value_score + - ceg: x_private:chip + use: display-only + type: string + example: "a" + renders: "a chip" + tag: "proposed:row_chip" +``` + +```yaml csd:states +populated: {tag: thing_list} +empty: {tag: thing_empty} +loading: {renders: "a spinner, no tag"} +error: {tag: "proposed:thing_error"} +``` +""" + + +def _csd_root(tmp_path: Path, text: str = CSD_TEXT, name: str = "CSD-900-test.md") -> Path: + root = tmp_path / "csd" + root.mkdir(exist_ok=True) + (root / name).write_text(text, encoding="utf-8") + return root + + +def _flow(tmp_path: Path, body: str, name: str = "f.yaml") -> Path: + p = tmp_path / name + p.write_text(textwrap.dedent(body), encoding="utf-8") + return p + + +GOOD = """\ + flow: thing + csd: CSD-900 + client: ">=0.5.224" + steps: + - step_id: land + title: lands + requires: {screen: Thing} + expect: + visible: [thing_list] +""" + + +# ── loading ───────────────────────────────────────────────────────────────── + +def test_a_good_flow_loads_and_carries_its_csds_maps(tmp_path): + spec = FlowSpec.load(_flow(tmp_path, GOOD), csd_root=_csd_root(tmp_path)) + assert spec.csd_id == "CSD-900" + assert spec.csd.field_tags["x_private:score"] == "value_score" + assert spec.csd.state_tags == {"populated": "thing_list", "empty": "thing_empty"} + assert "row_chip" in spec.csd.proposed and "thing_error" in spec.csd.proposed + + +def test_an_unknown_top_level_key_is_a_load_error(tmp_path): + p = _flow(tmp_path, GOOD + " csd_id: CSD-900\n") + with pytest.raises(SpecError, match="unknown key"): + FlowSpec.load(p, csd_root=_csd_root(tmp_path)) + + +def test_a_csd_that_does_not_exist_is_a_load_error_not_a_skip(tmp_path): + p = _flow(tmp_path, GOOD.replace("CSD-900", "CSD-999")) + with pytest.raises(SpecError, match="does not exist"): + FlowSpec.load(p, csd_root=_csd_root(tmp_path)) + + +def test_a_malformed_csd_id_is_a_load_error(tmp_path): + p = _flow(tmp_path, GOOD.replace("CSD-900", "people")) + with pytest.raises(SpecError, match="not a CSD id"): + FlowSpec.load(p, csd_root=_csd_root(tmp_path)) + + +def test_a_csd_whose_blocks_do_not_parse_is_a_load_error(tmp_path): + broken = CSD_TEXT.replace("populated: {tag: thing_list}", "populated: {tag: [unclosed") + with pytest.raises(SpecError, match="does not parse"): + FlowSpec.load(_flow(tmp_path, GOOD), csd_root=_csd_root(tmp_path, broken)) + + +def test_a_document_with_no_stage_block_is_not_a_csd(tmp_path): + prose = "# CSD-900\n\nJust prose, no typed blocks.\n" + with pytest.raises(SpecError, match="not a CSD/3 document"): + FlowSpec.load(_flow(tmp_path, GOOD), csd_root=_csd_root(tmp_path, prose)) + + +def test_a_flow_in_this_repo_must_name_its_csd(tmp_path): + no_csd = GOOD.replace(" csd: CSD-900\n", "") + d = tmp_path / "flows" + d.mkdir() + _flow(d, no_csd) + with pytest.raises(SpecError, match="names no `csd:`"): + run_flows.load_flows([d], csd_root=_csd_root(tmp_path)) + + +def test_an_empty_flows_directory_is_a_load_error(tmp_path): + d = tmp_path / "flows" + d.mkdir() + with pytest.raises(SpecError, match="no flows found"): + run_flows.load_flows([d]) + + +@pytest.mark.parametrize("where", [ + " expect:\n visible: [row_chip]\n", + " expect:\n absent: [row_chip]\n", + " expect:\n text: {row_chip: a}\n", + " do:\n - click: row_chip\n", + " requires:\n visible: [row_chip]\n expect:\n visible: [thing_list]\n", +]) +def test_a_proposed_tag_named_anywhere_in_a_flow_is_a_load_error(tmp_path, where): + body = GOOD.split(" - step_id")[0] + " - step_id: s\n title: t\n" + where + with pytest.raises(SpecError, match="proposed"): + FlowSpec.load(_flow(tmp_path, body), csd_root=_csd_root(tmp_path)) + + +def test_the_real_csd_005_refuses_its_proposed_trust_chip(tmp_path): + """Against the shipped document, not a fixture: `contacts_row_trust` is + `proposed:` in CSD-005, so no flow may assert it yet.""" + body = """\ + flow: p + csd: CSD-005 + steps: + - step_id: s + title: t + expect: + visible: [contacts_row_trust] + """ + with pytest.raises(SpecError, match="contacts_row_trust.*proposed"): + FlowSpec.load(_flow(tmp_path, body)) + + +def test_a_state_whose_tag_is_proposed_is_a_load_error(tmp_path): + body = GOOD.replace("visible: [thing_list]", "state: error") + with pytest.raises(SpecError, match="state: error.*proposed"): + FlowSpec.load(_flow(tmp_path, body), csd_root=_csd_root(tmp_path)) + + +def test_a_state_the_csd_gives_no_tag_is_a_load_error(tmp_path): + body = GOOD.replace("visible: [thing_list]", "state: loading") + with pytest.raises(SpecError, match="names no tag"): + FlowSpec.load(_flow(tmp_path, body), csd_root=_csd_root(tmp_path)) + + +def test_a_relation_over_a_field_the_csd_does_not_show_is_a_load_error(tmp_path): + body = GOOD.replace( + "visible: [thing_list]", + "relation: {left: 'x_private:score', op: eq, right: 'x_private:nope'}") + with pytest.raises(SpecError, match="not a field"): + FlowSpec.load(_flow(tmp_path, body), csd_root=_csd_root(tmp_path)) + + +def test_a_flow_without_csd_still_loads_through_flow_spec_alone(tmp_path): + """The delta is additive: an upstream-shaped flow (no `csd:`) loads as before.""" + spec = FlowSpec.load(_flow(tmp_path, GOOD.replace(" csd: CSD-900\n", ""))) + assert spec.csd is None + + +# ── the seeded flows ──────────────────────────────────────────────────────── + +def test_every_flow_in_the_repo_loads_against_its_real_csd(): + specs = run_flows.load_flows([FLOWS]) + assert specs, "testing/flows is empty — the matrix would run nothing and pass" + assert all(s.csd is not None for s in specs) + + +def _client_literals() -> set[str]: + src = REPO / "client" / "shared" / "src" / "commonMain" + found: set[str] = set() + for kt in src.rglob("*.kt"): + found.update(re.findall(r'"([a-z][a-z0-9_]+)"', kt.read_text(encoding="utf-8"))) + return found + + +def test_every_tag_a_seeded_flow_names_exists_in_the_client(): + """A flow naming a tag the client does not carry fails as 'element not + found' on every leg. Checked here, at the keyboard, rather than there.""" + literals = _client_literals() + missing = [] + for spec in run_flows.load_flows([FLOWS]): + for step in spec.steps: + tags = [a.target for a in step.do] + for cond in (step.requires, step.expect): + tags += cond.visible + cond.absent + list(cond.text) + if cond.state: + tags.append(spec.csd.state_tags[cond.state]) + missing += [f"{spec.flow}/{step.step_id}: {t}" for t in tags if t not in literals] + assert not missing, f"tags no client source carries: {missing}" + + +# ── the runner, over a fake helper ────────────────────────────────────────── + +@dataclass +class _El: + test_tag: str + text: Optional[str] = "" + visible: Optional[bool] = True + width: int = 10 + height: int = 10 + + +class FakeHelper: + """`/tree` as a dict. Records every call so a refusal can be shown to have + driven NOTHING.""" + + def __init__(self, screen: str, tags: Dict[str, str]): + self.screen = screen + self.els = {t: _El(t, txt) for t, txt in tags.items()} + self.calls: list[str] = [] + + async def get_elements(self): + self.calls.append("tree") + return list(self.els.values()) + + async def get_element(self, tag): + self.calls.append(f"get {tag}") + return self.els.get(tag) + + async def get_screen(self): + self.calls.append("screen") + return self.screen + + async def is_element_visible(self, tag): + return tag in self.els + + async def scroll_into_view(self, tag): + return tag in self.els + + async def click(self, tag, timeout=2000): + self.calls.append(f"click {tag}") + return tag in self.els + + async def input_text(self, tag, text): + self.calls.append(f"input {tag}") + return tag in self.els + + async def wait_for_element(self, tag, timeout=2000): + return tag in self.els + + +def _run(spec, helper, version="0.5.224"): + return asyncio.run(run_flows.run_one(spec, helper, client_version=version, start_timeout=0)) + + +def _spec(tmp_path, body=GOOD): + return FlowSpec.load(_flow(tmp_path, body), csd_root=_csd_root(tmp_path)) + + +def test_a_flow_whose_expects_hold_passes(tmp_path): + out = _run(_spec(tmp_path), FakeHelper("Thing", {"thing_list": ""})) + assert out.status == run_flows.PASS + assert run_flows.leg_ok([out]) + + +def test_a_failing_expect_fails_the_flow_and_the_leg(tmp_path): + out = _run(_spec(tmp_path), FakeHelper("Thing", {"something_else": ""})) + assert out.status == run_flows.FAIL + assert "thing_list" in out.detail and "expect" in out.detail + assert out.steps and out.steps[-1]["status"] == "fail" + assert not run_flows.leg_ok([out]) + + +def test_a_flow_that_never_reaches_its_first_screen_cannot_start_and_reddens_the_leg(tmp_path): + out = _run(_spec(tmp_path), FakeHelper("Login", {"thing_list": ""})) + assert out.status == run_flows.CANNOT_START + assert "Thing" in out.detail + assert not run_flows.leg_ok([out]), "a flow that never ran must not leave the leg green" + + +@pytest.mark.parametrize("floor,version", [ + (">=9.9.9", "0.5.224"), + (">0.5.224", "0.5.224"), + ("unreleased", "0.5.224"), +]) +def test_a_flow_above_its_floor_is_refused_neither_passed_nor_failed(tmp_path, floor, version): + spec = _spec(tmp_path, GOOD.replace('">=0.5.224"', f'"{floor}"')) + helper = FakeHelper("Thing", {"thing_list": ""}) + out = _run(spec, helper, version) + assert out.status == run_flows.REFUSED + assert helper.calls == [], "a refused flow must drive nothing" + assert run_flows.leg_ok([out]), "refused is not a failure" + assert "refused" in run_flows.summary([out]) + + +def test_a_met_floor_is_not_refused(tmp_path): + out = _run(_spec(tmp_path), FakeHelper("Thing", {"thing_list": ""}), "0.5.225+preview.gabc") + assert out.status == run_flows.PASS + + +def test_one_red_flow_among_green_ones_reddens_the_leg(tmp_path): + ok = run_flows.FlowOutcome("a", "CSD-900", run_flows.PASS) + refused = run_flows.FlowOutcome("b", "CSD-900", run_flows.REFUSED) + bad = run_flows.FlowOutcome("c", "CSD-900", run_flows.FAIL) + assert run_flows.leg_ok([ok, refused]) + assert not run_flows.leg_ok([ok, refused, bad]) + + +# ── `state:` is checked now, against the CSD's `states:` ──────────────────── + +STATE_FLOW = GOOD.replace("visible: [thing_list]", "state: empty") + + +def test_state_holds_when_its_tag_is_on_screen_and_no_other_states_is(tmp_path): + out = _run(_spec(tmp_path, STATE_FLOW), FakeHelper("Thing", {"thing_empty": ""})) + assert out.status == run_flows.PASS + + +def test_state_fails_when_its_tag_is_not_on_screen(tmp_path): + out = _run(_spec(tmp_path, STATE_FLOW), FakeHelper("Thing", {"other": ""})) + assert out.status == run_flows.FAIL and "thing_empty" in out.detail + + +def test_state_fails_when_another_states_tag_is_also_on_screen(tmp_path): + """Empty and populated at once is not 'empty' — the point of the map.""" + out = _run(_spec(tmp_path, STATE_FLOW), + FakeHelper("Thing", {"thing_empty": "", "thing_list": ""})) + assert out.status == run_flows.FAIL and "populated" in out.detail + + +def test_state_with_no_csd_map_fails_rather_than_passing(tmp_path): + """Upstream's `state:` was a no-op. Unanchored, it now refuses to be green.""" + from testing.gate.flow_spec import FlowRunner + spec = FlowSpec.load(_flow(tmp_path, STATE_FLOW.replace(" csd: CSD-900\n", ""))) + runner = FlowRunner(FakeHelper("Thing", {"thing_empty": ""})) + assert asyncio.run(runner.run(spec)) is False + assert "cannot be checked" in runner.results[-1].detail + + +# ── the workflow runs them on every leg ───────────────────────────────────── + +def test_every_leg_of_the_matrix_runs_the_flows(): + wf = yaml.safe_load((REPO / ".github" / "workflows" / "five-platform-live-qa.yml").read_text()) + legs = 0 + for job in wf["jobs"].values(): + for step in job.get("steps", []): + body = str(step.get("run", "")) + str((step.get("with") or {}).get("script", "")) + body = body.replace("\\\n", " ") # one logical command per line + for call in re.findall(r"testing\.gate\.run_platform[^;\n]*", body): + legs += 1 + assert "--flows testing/flows" in call, f"a leg runs the smoke walk without flows: {call}" + assert legs == 5, f"expected five run_platform legs, found {legs}" + + +# ── run_platform: flows ride the smoke walk, never ahead of it ────────────── + +def test_a_flow_that_does_not_load_stops_the_leg_before_anything_boots(tmp_path, monkeypatch): + from testing.gate import run_platform + bad = tmp_path / "flows" + bad.mkdir() + _flow(bad, GOOD.replace("CSD-900", "CSD-999")) + booted = [] + monkeypatch.setattr(run_platform, "plan_for", lambda a: booted.append(a)) + report = tmp_path / "r.json" + monkeypatch.setattr("sys.argv", ["run_platform", "--platform", "desktop", "--jar", "x.jar", + "--shots", str(tmp_path / "s"), "--report", str(report), + "--flows", str(bad)]) + assert run_platform.main() == 1 + assert booted == [], "a spec error must be found before the app is brought up" + got = yaml.safe_load(report.read_text()) + assert got["steps"][0]["name"] == "flows-load" and not got["ok"] + + +def test_flows_do_not_run_after_a_failed_smoke_walk(tmp_path): + from types import SimpleNamespace + from testing.gate import run_platform + rep = run_platform.Report(platform="desktop") + rep.add("ui-composed", False, "never composed") + run_platform.flows(None, rep, [_spec(tmp_path)], SimpleNamespace(), None) + assert rep.steps[-1].name == "flows" and not rep.steps[-1].ok + assert "not run" in rep.steps[-1].detail From b3a61da731ae6c5f271f1e265f64daf1746e0046 Mon Sep 17 00:00:00 2001 From: Eric Moore Date: Fri, 25 Sep 2026 18:48:14 -0500 Subject: [PATCH 02/19] fix(testing): the session fixture waits between fields on iOS and names what's on screen when a step won't advance The first iOS run stopped at "wizard did not advance past 'you'": the four fields were typed back to back, and iOS needs ~2 s per field for the value to reach the ViewModel's StateFlow (client/CLAUDE.md). A step that still won't advance now reports the tags on screen, so a required field the fixture doesn't fill shows in the failure. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01XXpYDXz2XUwUCG1ePtmVMK --- testing/gate/session_fixture.py | 11 ++++++++++- 1 file changed, 10 insertions(+), 1 deletion(-) diff --git a/testing/gate/session_fixture.py b/testing/gate/session_fixture.py index 8f52d205..11811117 100644 --- a/testing/gate/session_fixture.py +++ b/testing/gate/session_fixture.py @@ -98,6 +98,11 @@ def run_setup(drv: TestAutomationServer, username: str, password: str, drv.input(tag, value) except DriverError as e: raise SessionUnavailable(f"wizard: {tag} would not accept input ({e})") from e + # iOS needs ~2 s between fields for the value to reach the ViewModel's + # StateFlow (client/CLAUDE.md, "Important iOS notes"). Without it the + # fields read empty and Next never enables: the first iOS run of this + # fixture stopped at "wizard did not advance past 'you'". + time.sleep(2.0) drv.click("age_band_adult") # Advance until the claim takes over. Bounded: a wizard that stops advancing @@ -122,7 +127,11 @@ def run_setup(drv: TestAutomationServer, username: str, password: str, drv.click(nxt) time.sleep(2.0) if (drv.screen(), _active_step(drv)) == before and "setup_ownership_claimed" not in _tags(drv): - raise SessionUnavailable(f"wizard did not advance past {before[1]!r}") + # Say what was on screen: a required field the fixture doesn't fill + # (the with-AI wizard asks for more than the node one) shows up here. + raise SessionUnavailable( + f"wizard did not advance past {before[1]!r}; on screen: {sorted(_tags(drv))}" + ) # The claim has no button; it completes and the app returns to Login. if not _settle(drv, "Login", timeout=180): From 0e09159b80f3ac1a39f0852ab48719b97c91d10e Mon Sep 17 00:00:00 2001 From: Eric Moore Date: Fri, 25 Sep 2026 19:28:49 -0500 Subject: [PATCH 03/19] feat(testing): the flow runner walks to a flow's first screen; the tag check sees interpolated tags Sign-in lands on Contacts, so a flow for any other surface ended cannot-start. Before step one, run_one now walks nav_map's derived hop for the flow's first `requires: screen:` on this build (node or agent, from /state). It waits for each tag, then clicks it. - A missing hop tag is cannot-start and names the tag and its position. - A screen with no hop that is not flow-only is cannot-start: "no nav hop for Screen.X". - A flow-only screen (screen_atlas.flow_only) is waited for, not walked to. test_every_tag_a_seeded_flow_names_exists_in_the_client read only whole literals, so interpolated tags ("age_band_$token", "trace_consent_$token", "radio_cohort_$value", "chk_duty_box_$verb") looked absent. A tag now counts if it equals a literal or starts with the pre-`$` prefix of an interpolated one. Only prefixes with two segments count, so "btn_$x" cannot vouch for every button. Every new rule was mutation-checked red. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01XXpYDXz2XUwUCG1ePtmVMK --- testing/flows/README.md | 14 +++-- testing/gate/run_flows.py | 84 ++++++++++++++++++++++++-- testing/test_flows.py | 124 +++++++++++++++++++++++++++++++++++--- 3 files changed, 204 insertions(+), 18 deletions(-) diff --git a/testing/flows/README.md b/testing/flows/README.md index a8807ff2..7b97996b 100644 --- a/testing/flows/README.md +++ b/testing/flows/README.md @@ -72,7 +72,7 @@ a string in the client's `commonMain` source. |---|---|---| | `pass` | every step held | green | | `refused` | the `client:` floor is above the client under test (`>=X`, `>X`, `unreleased`) | green — reported, not passed | -| `cannot-start` | floor met, but the first step's `requires` never held | **red** | +| `cannot-start` | floor met, but the flow never reached its first screen (no hop, a hop tag missing, or the first `requires` never held) | **red** | | `fail` | it started and a step broke | **red** | `cannot-start` is red on purpose. The floor is how a flow waits for a surface @@ -129,10 +129,14 @@ bring-up per leg and one session. ## What this does not do yet -- **No navigation.** A flow's first step names its screen and the runner waits - for it; it does not walk the sidebar there. Every flow today starts where - sign-in lands (`Contacts`). A flow for another surface needs `nav_map` to drive - the hop — the CSD names the surface, and the hop is derived, never written. +- **Navigation is to the first screen only.** Before step one the runner walks + the hop `testing/gate/nav_map.py` derives for the flow's first + `requires: screen:` on this build (node or agent tree, from `/state`), + waiting for each tag before clicking it. A missing hop tag is `cannot-start` + naming the tag; a screen with no hop that is not flow-only is `cannot-start` + with "no nav hop for Screen.X"; a flow-only screen (pre-login, wizards, + leaves) is waited for, not walked to. Hops between later steps are the flow's + own `do:` clicks. - **Only what a bare node can show.** The matrix stands up one node with no contacts, no agent and no peers, so CSD-005's populated list and receipt sheet are not driven here. diff --git a/testing/gate/run_flows.py b/testing/gate/run_flows.py index 5dea6803..ac91d795 100644 --- a/testing/gate/run_flows.py +++ b/testing/gate/run_flows.py @@ -16,8 +16,15 @@ refused the flow's `client:` floor is above the client under test. The flow cannot start HERE, and says why; it is neither passed nor failed, and it does not redden the leg - cannot-start the floor is met, but the flow's first `requires` never held — - the client never reached the screen the flow starts on + cannot-start the floor is met, but the flow never reached its first screen: + nav_map has no hop to it on this build, a hop tag was missing + mid-walk, or the first `requires` still did not hold + +THE RUNNER WALKS TO THE FIRST SCREEN. Sign-in lands on Contacts; a flow for any +other surface starts elsewhere. Before step one, the runner clicks the hop +`nav_map` derives for the flow's first `requires: screen:` (circle, tab, row), +waiting for each tag. The flow never encodes the hop (FSD/CSD_STANDARD.md §5). +Flow-only screens (pre-login, wizards, leaves) have no hop and are waited for. fail it started and a step broke `cannot-start` REDDENS THE LEG, deliberately, and differs from CIRISAgent's gate @@ -109,16 +116,68 @@ async def _settle_on(helper, screen: str, timeout: float, poll: float = 1.0) -> return cur +async def navigate(helper, screen: str, chain: Sequence[str], *, hop_timeout: float = 20.0, + arrive_timeout: float = 20.0) -> Optional[str]: + """Walk `chain` (nav_map's derived hop) to `screen`. None on arrival, else + the reason — naming the hop tag that was missing, because "could not reach + Screen.X" alone sends the reader to the wrong end of the chain. + + Each tag is WAITED for before it is clicked: a circle's tabs compose after + the circle is chosen, and clicking before they exist is a race, not a test. + """ + for i, tag in enumerate(chain, 1): + where = f"hop {i} of {len(chain)} ({' -> '.join(chain)})" + if not await helper.wait_for_element(tag, timeout=int(hop_timeout * 1000)): + return (f"navigation to Screen.{screen}: hop tag {tag!r} never appeared, {where}; " + f"on {await helper.get_screen()!r}") + if not await helper.click(tag, timeout=int(hop_timeout * 1000)): + return f"navigation to Screen.{screen}: clicking hop tag {tag!r} failed, {where}" + got = await _settle_on(helper, screen, arrive_timeout) + if got != screen: + return (f"navigation to Screen.{screen}: walked {' -> '.join(chain)} and landed on " + f"{got!r}") + return None + + +def nav_hops(has_agent: bool) -> tuple[dict, set]: + """(Screen -> hop, flow-only screens) for this build, from the client source.""" + from testing.gate import nav_map, screen_atlas # noqa: PLC0415 + return nav_map.build(has_agent=has_agent), screen_atlas.flow_only() + + async def run_one(spec: FlowSpec, helper, *, platform=None, artifacts: Optional[Path] = None, - client_version: Optional[str] = None, start_timeout: float = 30.0) -> FlowOutcome: + client_version: Optional[str] = None, start_timeout: float = 30.0, + hops: Optional[dict] = None, flow_only: frozenset | set = frozenset()) -> FlowOutcome: + """Run one flow. With `hops` (nav_map's Screen -> chain), the runner first + WALKS to the flow's starting screen; without, it only waits for it.""" refusal = check_client_floor(spec.client_floor, client_version) if refusal: print(f"\n FLOW {spec.flow} ({spec.csd_id}) — REFUSED by its floor\n {refusal}") return FlowOutcome(spec.flow, spec.csd_id, REFUSED, refusal) start = spec.steps[0].requires.screen - if start: + if start and hops is None: await _settle_on(helper, start, start_timeout) + elif start: + # A landing still composing is not a flow on the wrong screen: give the + # client a moment before deciding to walk anywhere. + cur = await _settle_on(helper, start, min(start_timeout, 5.0)) + err = None + if cur != start: + if start in hops: + print(f"\n FLOW {spec.flow} — walking to Screen.{start}: {' -> '.join(hops[start])}") + err = await navigate(helper, start, hops[start], + hop_timeout=start_timeout, arrive_timeout=start_timeout) + elif start in flow_only: + # Pre-login, wizards, leaves: nothing in the shell leads there, + # so the flow must already be on it. Wait, then let `requires` judge. + await _settle_on(helper, start, start_timeout) + else: + err = (f"no nav hop for Screen.{start} on this build, and it is not a " + f"flow-only screen (on {cur!r})") + if err: + print(f"\n FLOW {spec.flow} ({spec.csd_id}) — CANNOT START\n {err}") + return FlowOutcome(spec.flow, spec.csd_id, CANNOT_START, err) runner = FlowRunner(helper, platform=platform, artifacts=artifacts) try: @@ -171,7 +230,7 @@ def sign_in(drv, username: str, password: str) -> str: def run_all(specs: Sequence[FlowSpec], drv, *, platform=None, artifacts: Optional[Path] = None, client_version: Optional[str] = None, username: str = "qaadmin", password: str = "QaAdmin!2345", establish_session: bool = True, - helper: Any = None) -> List[FlowOutcome]: + helper: Any = None, navigate_to_start: bool = True) -> List[FlowOutcome]: """Run every flow in order against one live client. Signs in once, only if some flow will actually run.""" from testing.gate.flow_helper import SyncFlowHelper # noqa: PLC0415 @@ -182,9 +241,22 @@ def run_all(specs: Sequence[FlowSpec], drv, *, platform=None, artifacts: Optiona landed = sign_in(drv, username, password) print(f" session: signed in, on {landed!r}") + hops, flow_only = None, frozenset() + if runnable and navigate_to_start: + # The build decides the tree: a node client has no agentOnly rows, so a + # hop derived for the agent build would click tags that are not there. + mode = "" + if drv is not None: + try: + mode = str(drv.state().get("clientMode", "")) + except Exception: # noqa: BLE001 — unknown mode: the node tree, the subset + mode = "" + hops, flow_only = nav_hops(has_agent=mode.upper() == "AGENT") + async def go() -> List[FlowOutcome]: return [await run_one(s, helper, platform=platform, artifacts=artifacts, - client_version=client_version) for s in specs] + client_version=client_version, hops=hops, + flow_only=flow_only) for s in specs] outcomes = asyncio.run(go()) print(f"\n flows: {summary(outcomes)}") diff --git a/testing/test_flows.py b/testing/test_flows.py index a324f4b0..5ea635c8 100644 --- a/testing/test_flows.py +++ b/testing/test_flows.py @@ -203,18 +203,47 @@ def test_every_flow_in_the_repo_loads_against_its_real_csd(): assert all(s.csd is not None for s in specs) -def _client_literals() -> set[str]: +def _client_tag_strings() -> tuple[set[str], set[str]]: + """(whole tag literals, prefixes of interpolated tags) in commonMain. + + `"age_band_$token"` builds `age_band_adult`, so a literal-only read calls a + real tag missing. A prefix counts only if it has two segments + (`age_band_`, not `btn_`): `"btn_$x"` would otherwise vouch for every + button a flow could ever name, which is no check at all. + """ src = REPO / "client" / "shared" / "src" / "commonMain" - found: set[str] = set() + literals: set[str] = set() + prefixes: set[str] = set() for kt in src.rglob("*.kt"): - found.update(re.findall(r'"([a-z][a-z0-9_]+)"', kt.read_text(encoding="utf-8"))) - return found + for body, end in re.findall(r'"([a-z][a-z0-9_]*)(["$])', kt.read_text(encoding="utf-8")): + if end == '"': + literals.add(body) + elif "_" in body.rstrip("_"): + prefixes.add(body) + return literals, prefixes + + +def client_carries(tag: str, literals: set[str], prefixes: set[str]) -> bool: + return tag in literals or any(tag.startswith(p) for p in prefixes) + + +@pytest.mark.parametrize("tag,carried", [ + ("age_band_adult", True), # "age_band_$token" SetupScreen.kt + ("trace_consent_yes", True), # "trace_consent_$token" SetupScreen.kt + ("radio_cohort_family", True), # "radio_cohort_$value" ClaimNodeScreen.kt + ("chk_duty_box_accept", True), # "chk_duty_box_$verb" DutyConferralScreen.kt + ("opt_run_with_ai", True), # a whole literal still matches + ("contacts_no_such_tag", False), + ("btn_no_such_button", False), # a one-segment prefix vouches for nothing +]) +def test_the_client_tag_check_sees_interpolated_tags_and_nothing_else(tag, carried): + assert client_carries(tag, *_client_tag_strings()) is carried def test_every_tag_a_seeded_flow_names_exists_in_the_client(): """A flow naming a tag the client does not carry fails as 'element not found' on every leg. Checked here, at the keyboard, rather than there.""" - literals = _client_literals() + literals, prefixes = _client_tag_strings() missing = [] for spec in run_flows.load_flows([FLOWS]): for step in spec.steps: @@ -223,7 +252,8 @@ def test_every_tag_a_seeded_flow_names_exists_in_the_client(): tags += cond.visible + cond.absent + list(cond.text) if cond.state: tags.append(spec.csd.state_tags[cond.state]) - missing += [f"{spec.flow}/{step.step_id}: {t}" for t in tags if t not in literals] + missing += [f"{spec.flow}/{step.step_id}: {t}" for t in tags + if not client_carries(t, literals, prefixes)] assert not missing, f"tags no client source carries: {missing}" @@ -246,6 +276,7 @@ def __init__(self, screen: str, tags: Dict[str, str]): self.screen = screen self.els = {t: _El(t, txt) for t, txt in tags.items()} self.calls: list[str] = [] + self.leads: dict = {} async def get_elements(self): self.calls.append("tree") @@ -267,7 +298,13 @@ async def scroll_into_view(self, tag): async def click(self, tag, timeout=2000): self.calls.append(f"click {tag}") - return tag in self.els + if tag not in self.els: + return False + # A click can move the app: `leads` maps a tag to (screen, tags now shown). + if tag in self.leads: + self.screen, shown = self.leads[tag] + self.els = {t: _El(t, "") for t in shown} + return True async def input_text(self, tag, text): self.calls.append(f"input {tag}") @@ -407,3 +444,76 @@ def test_flows_do_not_run_after_a_failed_smoke_walk(tmp_path): run_platform.flows(None, rep, [_spec(tmp_path)], SimpleNamespace(), None) assert rep.steps[-1].name == "flows" and not rep.steps[-1].ok assert "not run" in rep.steps[-1].detail + + +# ── navigation: the runner walks to a flow's first screen ─────────────────── + +HOPS = {"Thing": ["circle_x", "tab_y", "nav_thing"]} + + +def _walkable(missing: str = "") -> FakeHelper: + """Lands on Contacts; circle_x -> tab_y -> nav_thing reaches Thing.""" + h = FakeHelper("Contacts", {"circle_x": ""}) + h.leads = { + "circle_x": ("CircleTab", ["circle_x", "tab_y"]), + "tab_y": ("CircleTab", ["circle_x", "tab_y", "nav_thing"]), + "nav_thing": ("Thing", ["thing_list"]), + } + if missing: + for screen, shown in h.leads.values(): + if missing in shown: + shown.remove(missing) + return h + + +def _nav_run(spec, helper, hops=HOPS, flow_only=frozenset()): + return asyncio.run(run_flows.run_one(spec, helper, client_version="0.5.224", + start_timeout=0, hops=hops, flow_only=flow_only)) + + +def test_the_runner_walks_the_derived_hop_to_the_first_screen(tmp_path): + h = _walkable() + out = _nav_run(_spec(tmp_path), h) + assert out.status == run_flows.PASS, out.detail + assert [c for c in h.calls if c.startswith("click")] == [ + "click circle_x", "click tab_y", "click nav_thing"] + + +def test_a_missing_hop_tag_is_cannot_start_and_names_the_tag(tmp_path): + out = _nav_run(_spec(tmp_path), _walkable(missing="tab_y")) + assert out.status == run_flows.CANNOT_START + assert "'tab_y'" in out.detail and "hop 2 of 3" in out.detail + assert "never appeared" in out.detail, "waited for, not blindly clicked" + assert not run_flows.leg_ok([out]) + + +def test_a_screen_with_no_hop_that_is_not_flow_only_cannot_start(tmp_path): + h = _walkable() + out = _nav_run(_spec(tmp_path), h, hops={}) + assert out.status == run_flows.CANNOT_START + assert "no nav hop for Screen.Thing" in out.detail + assert not [c for c in h.calls if c.startswith("click")], "nothing to walk, nothing clicked" + + +def test_a_flow_only_screen_is_waited_for_not_walked_to(tmp_path): + """No hop exists, and that is not a defect: the flow must already be there.""" + h = FakeHelper("Thing", {"thing_list": ""}) + out = _nav_run(_spec(tmp_path), h, hops={}, flow_only={"Thing"}) + assert out.status == run_flows.PASS + elsewhere = FakeHelper("Contacts", {"thing_list": ""}) + out = _nav_run(_spec(tmp_path), elsewhere, hops={}, flow_only={"Thing"}) + assert out.status == run_flows.CANNOT_START and "Thing" in out.detail + assert "no nav hop" not in out.detail, "flow-only is not a missing hop" + + +def test_already_on_the_first_screen_walks_nothing(tmp_path): + h = FakeHelper("Thing", {"thing_list": ""}) + assert _nav_run(_spec(tmp_path), h).status == run_flows.PASS + assert not [c for c in h.calls if c.startswith("click")] + + +def test_the_real_nav_map_reaches_the_seeded_flows_first_screens(): + hops, flow_only = run_flows.nav_hops(has_agent=False) + for spec in run_flows.load_flows([FLOWS]): + start = spec.steps[0].requires.screen + assert start in hops or start in flow_only, f"{spec.flow}: no way to Screen.{start}" From 1c4e0e25a1057ac8add98b1d1bc422df3b8da9df Mon Sep 17 00:00:00 2001 From: Eric Moore Date: Mon, 28 Sep 2026 13:47:37 -0500 Subject: [PATCH 04/19] fix(gate): the session fixture waits for the wizard's Next to become clickable before clicking it On iOS the typed fields reach the ViewModel a beat late, so Next is on screen but disabled, and a disabled control has no click handler: the click was a bare 404 that named no step. Wait (bounded) for can_click, and if it never enables say which step and what was on screen. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_0155SkTGdbwnSnR6tWBqJvUM --- testing/gate/session_fixture.py | 23 +++++++++++++++++++++++ 1 file changed, 23 insertions(+) diff --git a/testing/gate/session_fixture.py b/testing/gate/session_fixture.py index 11811117..6ab622db 100644 --- a/testing/gate/session_fixture.py +++ b/testing/gate/session_fixture.py @@ -64,6 +64,19 @@ def _tags(drv: TestAutomationServer) -> set[str]: return {e.test_tag for e in drv.tree()} +def _wait_clickable(drv: TestAutomationServer, tag: str, timeout: float = 20.0) -> bool: + """True once `tag` reports can_click (or the server does not report it at all).""" + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + for e in drv.tree(): + if e.test_tag == tag: + if e.can_click is None or e.can_click: + return True + break + time.sleep(1.0) + return False + + def _settle(drv: TestAutomationServer, want: str, timeout: float = 90.0) -> bool: deadline = time.monotonic() + timeout while time.monotonic() < deadline: @@ -123,6 +136,16 @@ def run_setup(drv: TestAutomationServer, username: str, password: str, f"wizard step {_active_step(drv)!r} offers no advance control; " f"on screen: {sorted(tags)}" ) + # A disabled advance control is present but has no click handler (a + # `testableClickable(enabled = false)`), and clicking it is a 404 that + # says nothing about WHY. On iOS the fields reach the ViewModel a beat + # after they are typed, so Next enables late: wait for it, bounded, and + # if it never enables say which step and what was on screen. + if not _wait_clickable(drv, nxt, timeout=20.0): + raise SessionUnavailable( + f"wizard step {_active_step(drv)!r}: {nxt} never became clickable; " + f"on screen: {sorted(_tags(drv))}" + ) before = (drv.screen(), _active_step(drv)) drv.click(nxt) time.sleep(2.0) From 3c9344b28fd91676c541e765eb9ce9e0ed327432 Mon Sep 17 00:00:00 2001 From: Eric Moore Date: Mon, 28 Sep 2026 13:56:22 -0500 Subject: [PATCH 05/19] fix(gate): the session fixture lets the trace-consent answer land before Next, and retries once JOIN_FEDERATION advances only when traceConsentAnswered; a Next clicked in the same instant as the answer raced it on Windows and the step reported 'did not advance' with the question still on screen. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_0155SkTGdbwnSnR6tWBqJvUM --- testing/gate/session_fixture.py | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/testing/gate/session_fixture.py b/testing/gate/session_fixture.py index 6ab622db..47d69cca 100644 --- a/testing/gate/session_fixture.py +++ b/testing/gate/session_fixture.py @@ -129,6 +129,11 @@ def run_setup(drv: TestAutomationServer, username: str, password: str, # answer for the same reason age_band_adult is: the unrestricted path. if "trace_consent_yes" in tags: drv.click("trace_consent_yes") + # The step advances only once the answer has reached the ViewModel + # (`SetupState.canProceedFromCurrentStep`: JOIN_FEDERATION -> + # traceConsentAnswered). A Next clicked in the same instant as the + # answer raced it on Windows; give the answer a beat to land. + time.sleep(1.5) tags = _tags(drv) nxt = "btn_wizard_complete" if "btn_wizard_complete" in tags else "btn_next" if nxt not in tags: @@ -149,6 +154,14 @@ def run_setup(drv: TestAutomationServer, username: str, password: str, before = (drv.screen(), _active_step(drv)) drv.click(nxt) time.sleep(2.0) + if (drv.screen(), _active_step(drv)) == before and "setup_ownership_claimed" not in _tags(drv): + # One retry when the step's question is still on screen: the answer + # may not have landed before Next was clicked. + if "trace_consent_yes" in _tags(drv): + drv.click("trace_consent_yes") + time.sleep(2.0) + drv.click(nxt) + time.sleep(2.0) if (drv.screen(), _active_step(drv)) == before and "setup_ownership_claimed" not in _tags(drv): # Say what was on screen: a required field the fixture doesn't fill # (the with-AI wizard asks for more than the node one) shows up here. From 7f89e746654baebf4f7c9058bf2f14647bea59fa Mon Sep 17 00:00:00 2001 From: Eric Moore Date: Mon, 28 Sep 2026 14:43:59 -0500 Subject: [PATCH 06/19] =?UTF-8?q?fix(gate):=20a=20disabled=20advance=20con?= =?UTF-8?q?trol=20on=20iOS=20answers=20404,=20not=20canClick=3Dfalse=20?= =?UTF-8?q?=E2=80=94=20retry,=20bounded,=20then=20name=20the=20step?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The iOS automation tree omits canClick when false, so the clickable wait passes; the click then 404s with 'No click handler'. Treat that answer as 'not yet' for up to 30 s and, if it stays, say which step and what was on screen instead of a bare 404. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_0155SkTGdbwnSnR6tWBqJvUM --- testing/gate/session_fixture.py | 20 +++++++++++++++++++- 1 file changed, 19 insertions(+), 1 deletion(-) diff --git a/testing/gate/session_fixture.py b/testing/gate/session_fixture.py index 47d69cca..e983653b 100644 --- a/testing/gate/session_fixture.py +++ b/testing/gate/session_fixture.py @@ -152,7 +152,25 @@ def run_setup(drv: TestAutomationServer, username: str, password: str, f"on screen: {sorted(_tags(drv))}" ) before = (drv.screen(), _active_step(drv)) - drv.click(nxt) + # iOS's /tree omits `canClick` when it is false (defaults are not + # serialized), so `_wait_clickable` cannot see a disabled Next there. + # A disabled control answers the click with 404 "No click handler": + # treat that as "not yet", bounded, and name the step if it stays so. + clicked = False + for _ in range(15): + try: + drv.click(nxt) + clicked = True + break + except DriverError as e: + if "No click handler" not in str(e): + raise + time.sleep(2.0) + if not clicked: + raise SessionUnavailable( + f"wizard step {before[1]!r}: {nxt} stayed disabled for 30s; " + f"on screen: {sorted(_tags(drv))}" + ) time.sleep(2.0) if (drv.screen(), _active_step(drv)) == before and "setup_ownership_claimed" not in _tags(drv): # One retry when the step's question is still on screen: the answer From 4c94e33e9f881c00f53318dc840e14df96f3b9a6 Mon Sep 17 00:00:00 2001 From: Eric Moore Date: Mon, 28 Sep 2026 15:35:45 -0500 Subject: [PATCH 07/19] fix(ci): one pytest invocation for the flow and state-tag tests (a merge left the second on its own line, run as a command) Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_0155SkTGdbwnSnR6tWBqJvUM --- .github/workflows/build.yml | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index ac2c83e8..ab085428 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -111,8 +111,7 @@ jobs: python3 -m pip install --quiet pytest pyyaml python3 -m pytest testing/test_driver_rules.py testing/test_gate_vendoring.py \ testing/test_bringup.py testing/test_five_platform_workflow.py \ - testing/test_flows.py -q - testing/test_csd_state_tags.py -q + testing/test_flows.py testing/test_csd_state_tags.py -q - name: A published version must offer a universal wheel # Only for versions already on the index — a PR's VERSION is normally From 2c8c0ee3eff13e4a1a05df61b95e2b9cd38bc4bc Mon Sep 17 00:00:00 2001 From: Eric Moore Date: Mon, 28 Sep 2026 15:38:39 -0500 Subject: [PATCH 08/19] fix(gate): the session fixture fills the fed-ID label when the wizard asks for it On iOS the YOU step keeps Next disabled until the federation-ID label is valid (desktop mints the identity itself); the fixture never filled it, so the walk stopped at 'you'. The bounded wait added before named it. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_0155SkTGdbwnSnR6tWBqJvUM --- testing/gate/session_fixture.py | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/testing/gate/session_fixture.py b/testing/gate/session_fixture.py index e983653b..2bff62d9 100644 --- a/testing/gate/session_fixture.py +++ b/testing/gate/session_fixture.py @@ -116,6 +116,16 @@ def run_setup(drv: TestAutomationServer, username: str, password: str, # fields read empty and Next never enables: the first iOS run of this # fixture stopped at "wizard did not advance past 'you'". time.sleep(2.0) + # The fed-ID label: desktop mints/admits the identity itself, but a client + # that asks (iOS) keeps Next disabled until the label is valid + # (SetupState.canProceedFromCurrentStep: YOU -> fedIdOk; generic words + # like "me" are refused, so use the device name, which is specific). + if "input_fedid_label" in _tags(drv): + try: + drv.input("input_fedid_label", f"{device} gate identity") + except DriverError as e: + raise SessionUnavailable(f"wizard: input_fedid_label would not accept input ({e})") from e + time.sleep(2.0) drv.click("age_band_adult") # Advance until the claim takes over. Bounded: a wizard that stops advancing From 4acb3f762b1f840877e55cc5f1b4028244fd453c Mon Sep 17 00:00:00 2001 From: Eric Moore Date: Mon, 28 Sep 2026 16:23:47 -0500 Subject: [PATCH 09/19] test(gate): a stuck wizard step reports each field's value, text and capability Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_0155SkTGdbwnSnR6tWBqJvUM --- testing/gate/session_fixture.py | 16 +++++++++++++++- 1 file changed, 15 insertions(+), 1 deletion(-) diff --git a/testing/gate/session_fixture.py b/testing/gate/session_fixture.py index 2bff62d9..7a718a1a 100644 --- a/testing/gate/session_fixture.py +++ b/testing/gate/session_fixture.py @@ -77,6 +77,20 @@ def _wait_clickable(drv: TestAutomationServer, tag: str, timeout: float = 20.0) return False +def _field_report(drv: TestAutomationServer) -> str: + """What each tagged element on screen holds: input values (passwords by + length only), texts, and click/input capability — so a disabled Next + names the condition it is waiting on instead of just the tag list.""" + parts = [] + for e in drv.tree(): + val = getattr(e, "input_value", None) + if val is not None and "password" in e.test_tag: + val = f"<{len(val)} chars>" + txt = (e.text or "")[:60] + parts.append(f"{e.test_tag}[v={val!r} t={txt!r} c={e.can_click} i={e.can_input}]") + return "; ".join(sorted(parts)) + + def _settle(drv: TestAutomationServer, want: str, timeout: float = 90.0) -> bool: deadline = time.monotonic() + timeout while time.monotonic() < deadline: @@ -179,7 +193,7 @@ def run_setup(drv: TestAutomationServer, username: str, password: str, if not clicked: raise SessionUnavailable( f"wizard step {before[1]!r}: {nxt} stayed disabled for 30s; " - f"on screen: {sorted(_tags(drv))}" + f"fields: {_field_report(drv)}" ) time.sleep(2.0) if (drv.screen(), _active_step(drv)) == before and "setup_ownership_claimed" not in _tags(drv): From e646077a5c5d6ae797dccb546e947b39bcafdfbb Mon Sep 17 00:00:00 2001 From: Eric Moore Date: Mon, 28 Sep 2026 17:36:09 -0500 Subject: [PATCH 10/19] fix(test-automation): a failed /input answers 404 on iOS and Android, and the gate driver raises on success:false iOS answered a failed text input with HTTP 200 {success:false}; the driver checked only the status, so every field the session fixture typed on iOS silently took nothing and the wizard's Next stayed disabled. Desktop already answered 404. Android had the same 200. The driver now treats a success:false body as a failure whatever the status (test red on the old driver, green now). Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_0155SkTGdbwnSnR6tWBqJvUM --- client/VENDORING.md | 2 +- .../testing/TestAutomationServer.android.kt | 4 ++- .../testing/TestAutomationServer.ios.kt | 6 +++- testing/driver.py | 8 +++++- testing/test_driver_rules.py | 28 +++++++++++++++++++ 5 files changed, 44 insertions(+), 4 deletions(-) diff --git a/client/VENDORING.md b/client/VENDORING.md index 6917f885..77c48d1c 100644 --- a/client/VENDORING.md +++ b/client/VENDORING.md @@ -40,7 +40,7 @@ source is the pair a bisect wants: The tree's current recorded state — sha256-of-sha256s over every git-tracked file under `client/` except this one: -**state digest:** `aac7cd60a1bb719d5ed274e75b06928e0abd02c682c4960a75bd367ae3606d72` +**state digest:** `7a0ba528c6e055ca6f079b4dcb53ea950a2a8ec7f1d09eeb7b43866ee9aa93eb` `packaging/check_vendoring.py` asserts it on every push, and refuses any tracked file matching a §2 never-vendor class. **Any commit that touches diff --git a/client/shared/src/androidMain/kotlin/ai/ciris/mobile/shared/testing/TestAutomationServer.android.kt b/client/shared/src/androidMain/kotlin/ai/ciris/mobile/shared/testing/TestAutomationServer.android.kt index 30497731..3d131da8 100644 --- a/client/shared/src/androidMain/kotlin/ai/ciris/mobile/shared/testing/TestAutomationServer.android.kt +++ b/client/shared/src/androidMain/kotlin/ai/ciris/mobile/shared/testing/TestAutomationServer.android.kt @@ -94,7 +94,9 @@ class AndroidTestAutomationServer(private val port: Int = 9091) { // Input text to element post("/input") { val request = call.receive() - call.respond(TestAutomationHandler.handleInput(request)) + val resp = TestAutomationHandler.handleInput(request) + // A failed input is not a 200 (desktop answers 404; iOS now does too). + call.respond(if (resp.success) HttpStatusCode.OK else HttpStatusCode.NotFound, resp) } // Scroll the screen (recovery after an off-screen refusal) diff --git a/client/shared/src/iosMain/kotlin/ai/ciris/mobile/shared/testing/TestAutomationServer.ios.kt b/client/shared/src/iosMain/kotlin/ai/ciris/mobile/shared/testing/TestAutomationServer.ios.kt index 6cc5540f..e166d3bc 100644 --- a/client/shared/src/iosMain/kotlin/ai/ciris/mobile/shared/testing/TestAutomationServer.ios.kt +++ b/client/shared/src/iosMain/kotlin/ai/ciris/mobile/shared/testing/TestAutomationServer.ios.kt @@ -265,7 +265,11 @@ class IOSTestAutomationServer(private val port: Int = 9091) { } method == "POST" && path == "/input" -> { val req = json.decodeFromString(body) - 200 to json.encodeToString(TestAutomationHandler.handleInput(req)) + val resp = TestAutomationHandler.handleInput(req) + // A failed input is not a 200: desktop answers 404, and a + // harness that checks the status (the gate's driver did) + // otherwise believes it typed into a field that took nothing. + (if (resp.success) 200 else 404) to json.encodeToString(resp) } method == "POST" && path == "/wait" -> { val req = json.decodeFromString(body) diff --git a/testing/driver.py b/testing/driver.py index 2d383f77..02b4e184 100644 --- a/testing/driver.py +++ b/testing/driver.py @@ -138,9 +138,15 @@ def _call(self, method: str, route: str, body: dict | None = None) -> Any: if not raw.strip(): return None try: - return json.loads(raw) + parsed = json.loads(raw) except json.JSONDecodeError: return raw + # A body that says it failed has failed, whatever the HTTP status: the + # iOS server answered a failed /input with 200 until 0.5.225, and the + # walk then "typed" into fields that took nothing. + if isinstance(parsed, dict) and parsed.get("success") is False: + raise DriverError(f"{method} {route} -> {parsed.get('error') or parsed}") + return parsed # ---- reads -------------------------------------------------------- diff --git a/testing/test_driver_rules.py b/testing/test_driver_rules.py index c483c6a6..ad890566 100644 --- a/testing/test_driver_rules.py +++ b/testing/test_driver_rules.py @@ -206,3 +206,31 @@ def test_wait_for_ui_does_not_confuse_a_live_server_for_a_composed_app(server): assert drv.wait_for_ui(timeout=1.0) == 0, ( "a reachable automation server was mistaken for a composed app" ) + + +def test_a_body_that_says_it_failed_raises_even_on_http_200(): + """iOS answered a failed /input with 200 {"success": false}; the walk then + 'typed' into fields that took nothing (CSD flows on the matrix, #97).""" + import http.server, threading, json as _json + from testing.driver import TestAutomationServer, DriverError + + class H(http.server.BaseHTTPRequestHandler): + def do_POST(self): + n = int(self.headers.get("Content-Length", 0)); self.rfile.read(n) + body = _json.dumps({"success": False, "error": "no text sink is listening for input_x"}).encode() + self.send_response(200); self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(body))); self.end_headers(); self.wfile.write(body) + def log_message(self, *a): pass + + srv = http.server.HTTPServer(("127.0.0.1", 0), H) + t = threading.Thread(target=srv.serve_forever, daemon=True); t.start() + try: + drv = TestAutomationServer(base_url=f"http://127.0.0.1:{srv.server_address[1]}") + try: + drv.input("input_x", "hello", verify=False) + except DriverError as e: + assert "no text sink" in str(e) + else: + raise AssertionError("a success:false body was accepted as typed") + finally: + srv.shutdown() From 23c8ba16b810fce96a6d6e13c1033e1ab8cbad82 Mon Sep 17 00:00:00 2001 From: Eric Moore Date: Mon, 28 Sep 2026 18:13:36 -0500 Subject: [PATCH 11/19] fix(gate): the session fixture scrolls a wizard field into view before typing or clicking it On the iPhone the first-run wizard is taller than the screen: the account fields sit below the age band and the AI choice, and /input refuses a field the person could not see (CIRISClient#33). With #97's 404-on-failure fix the refusal finally surfaced. Scroll down in steps, bounded, on an off-screen refusal; everything else raises as before. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_0155SkTGdbwnSnR6tWBqJvUM --- testing/gate/session_fixture.py | 32 +++++++++++++++++++++++++++----- 1 file changed, 27 insertions(+), 5 deletions(-) diff --git a/testing/gate/session_fixture.py b/testing/gate/session_fixture.py index 7a718a1a..00f6e79a 100644 --- a/testing/gate/session_fixture.py +++ b/testing/gate/session_fixture.py @@ -77,6 +77,27 @@ def _wait_clickable(drv: TestAutomationServer, tag: str, timeout: float = 20.0) return False +def _reach(drv: TestAutomationServer, tag: str, act, tries: int = 8): + """Run `act()` against `tag`, scrolling it into view first when the app + says it is composed but off screen. A phone's first-run wizard is taller + than the screen: on the iPhone the account fields sit below the age band + and the AI choice, and /input and /click refuse what the person could not + see (CIRISClient#33). Scroll down a step at a time, bounded; anything else + raises as before.""" + for _ in range(tries): + try: + return act() + except DriverError as e: + if "off screen" not in str(e): + raise + try: + drv.scroll_to(tag, direction="down", amount=400) + except DriverError: + pass + time.sleep(0.8) + return act() + + def _field_report(drv: TestAutomationServer) -> str: """What each tagged element on screen holds: input values (passwords by length only), texts, and click/input capability — so a disabled Next @@ -122,7 +143,7 @@ def run_setup(drv: TestAutomationServer, username: str, password: str, ("input_password_confirm", password), ("input_device_name", device)): try: - drv.input(tag, value) + _reach(drv, tag, lambda t=tag, v=value: drv.input(t, v)) except DriverError as e: raise SessionUnavailable(f"wizard: {tag} would not accept input ({e})") from e # iOS needs ~2 s between fields for the value to reach the ViewModel's @@ -136,11 +157,12 @@ def run_setup(drv: TestAutomationServer, username: str, password: str, # like "me" are refused, so use the device name, which is specific). if "input_fedid_label" in _tags(drv): try: - drv.input("input_fedid_label", f"{device} gate identity") + _reach(drv, "input_fedid_label", + lambda: drv.input("input_fedid_label", f"{device} gate identity")) except DriverError as e: raise SessionUnavailable(f"wizard: input_fedid_label would not accept input ({e})") from e time.sleep(2.0) - drv.click("age_band_adult") + _reach(drv, "age_band_adult", lambda: drv.click("age_band_adult")) # Advance until the claim takes over. Bounded: a wizard that stops advancing # must say so rather than spin. @@ -152,7 +174,7 @@ def run_setup(drv: TestAutomationServer, username: str, password: str, # answered (no default, like the age band above). Yes is the fixture's # answer for the same reason age_band_adult is: the unrestricted path. if "trace_consent_yes" in tags: - drv.click("trace_consent_yes") + _reach(drv, "trace_consent_yes", lambda: drv.click("trace_consent_yes")) # The step advances only once the answer has reached the ViewModel # (`SetupState.canProceedFromCurrentStep`: JOIN_FEDERATION -> # traceConsentAnswered). A Next clicked in the same instant as the @@ -183,7 +205,7 @@ def run_setup(drv: TestAutomationServer, username: str, password: str, clicked = False for _ in range(15): try: - drv.click(nxt) + _reach(drv, nxt, lambda: drv.click(nxt)) clicked = True break except DriverError as e: From c25bcd88189b9be8a667e425ab9b39c8ea8cc2fd Mon Sep 17 00:00:00 2001 From: Eric Moore Date: Mon, 28 Sep 2026 18:54:49 -0500 Subject: [PATCH 12/19] fix(gate): scroll back up to a wizard element the earlier steps scrolled past Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_0155SkTGdbwnSnR6tWBqJvUM --- testing/gate/session_fixture.py | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/testing/gate/session_fixture.py b/testing/gate/session_fixture.py index 00f6e79a..52b4ee94 100644 --- a/testing/gate/session_fixture.py +++ b/testing/gate/session_fixture.py @@ -84,14 +84,16 @@ def _reach(drv: TestAutomationServer, tag: str, act, tries: int = 8): and the AI choice, and /input and /click refuse what the person could not see (CIRISClient#33). Scroll down a step at a time, bounded; anything else raises as before.""" - for _ in range(tries): + # Down first (the wizard fills top to bottom), then back up past the + # start: an element the earlier steps scrolled past sits ABOVE the fold. + for direction in ["down"] * (tries // 2) + ["up"] * tries: try: return act() except DriverError as e: if "off screen" not in str(e): raise try: - drv.scroll_to(tag, direction="down", amount=400) + drv.scroll_to(tag, direction=direction, amount=400) except DriverError: pass time.sleep(0.8) From ca4ed2c48b4a4a34cd41c10f8ba0558c9038b08a Mon Sep 17 00:00:00 2001 From: Eric Moore Date: Mon, 28 Sep 2026 19:40:53 -0500 Subject: [PATCH 13/19] test(gate): an off-screen wizard element that scrolling cannot reach reports what each scroll answered Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_0155SkTGdbwnSnR6tWBqJvUM --- testing/gate/session_fixture.py | 16 ++++++++++++---- 1 file changed, 12 insertions(+), 4 deletions(-) diff --git a/testing/gate/session_fixture.py b/testing/gate/session_fixture.py index 52b4ee94..d0707eb2 100644 --- a/testing/gate/session_fixture.py +++ b/testing/gate/session_fixture.py @@ -86,6 +86,7 @@ def _reach(drv: TestAutomationServer, tag: str, act, tries: int = 8): raises as before.""" # Down first (the wizard fills top to bottom), then back up past the # start: an element the earlier steps scrolled past sits ABOVE the fold. + notes: list[str] = [] for direction in ["down"] * (tries // 2) + ["up"] * tries: try: return act() @@ -93,11 +94,18 @@ def _reach(drv: TestAutomationServer, tag: str, act, tries: int = 8): if "off screen" not in str(e): raise try: - drv.scroll_to(tag, direction=direction, amount=400) - except DriverError: - pass + r = drv.scroll_to(tag, direction=direction, amount=400) + notes.append(f"{direction}: {(r or {}).get('error') or 'moved'}" + if isinstance(r, dict) else f"{direction}: moved") + except DriverError as se: + notes.append(f"{direction}: {str(se)[-160:]}") time.sleep(0.8) - return act() + try: + return act() + except DriverError as e: + # Say what the scrolls answered: "no overflow" means the wizard's own + # scrollable is not the one registered, which is a client defect. + raise DriverError(f"{e} | scrolls: {'; '.join(dict.fromkeys(notes))}") from None def _field_report(drv: TestAutomationServer) -> str: From b64f69efc3ba38906ce009665e898f0e38e2d929 Mon Sep 17 00:00:00 2001 From: Eric Moore Date: Mon, 28 Sep 2026 20:19:50 -0500 Subject: [PATCH 14/19] fix(gate): scroll to the bottom, then the top, checking after every step, before calling a wizard field unreachable A phone with the keyboard up has a small viewport; a fixed four steps down turned around before reaching the device-name field. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_0155SkTGdbwnSnR6tWBqJvUM --- testing/gate/session_fixture.py | 29 ++++++++++++++++++----------- 1 file changed, 18 insertions(+), 11 deletions(-) diff --git a/testing/gate/session_fixture.py b/testing/gate/session_fixture.py index d0707eb2..73d2e8fd 100644 --- a/testing/gate/session_fixture.py +++ b/testing/gate/session_fixture.py @@ -87,19 +87,26 @@ def _reach(drv: TestAutomationServer, tag: str, act, tries: int = 8): # Down first (the wizard fills top to bottom), then back up past the # start: an element the earlier steps scrolled past sits ABOVE the fold. notes: list[str] = [] - for direction in ["down"] * (tries // 2) + ["up"] * tries: - try: - return act() - except DriverError as e: - if "off screen" not in str(e): - raise + # Down until the screen says it is at the bottom, then up until the top, + # trying the act after every step. A phone with the keyboard up has a + # small viewport, so a fixed number of steps can turn around before it + # ever reaches the last field. + for direction in ("down", "up"): + for _ in range(tries * 3): + try: + return act() + except DriverError as e: + if "off screen" not in str(e): + raise try: - r = drv.scroll_to(tag, direction=direction, amount=400) - notes.append(f"{direction}: {(r or {}).get('error') or 'moved'}" - if isinstance(r, dict) else f"{direction}: moved") + r = drv.scroll_to(tag, direction=direction, amount=300) + msg = (r or {}).get("error") if isinstance(r, dict) else None except DriverError as se: - notes.append(f"{direction}: {str(se)[-160:]}") - time.sleep(0.8) + msg = str(se)[-160:] + notes.append(f"{direction}: {msg or 'moved'}") + time.sleep(0.6) + if msg and ("already at the" in msg or "NO overflow" in msg): + break try: return act() except DriverError as e: From 498aec62e5ca4cdb76ea5e5a828280e0595d3ee5 Mon Sep 17 00:00:00 2001 From: Eric Moore Date: Mon, 28 Sep 2026 21:02:30 -0500 Subject: [PATCH 15/19] fix(gate): the session fixture chooses 'run without AI' when the wizard asks The legs run against a bare node with no LLM; iOS asks the question and the walk stopped at the AI step, whose Next waits for a usable LLM choice. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_0155SkTGdbwnSnR6tWBqJvUM --- testing/gate/session_fixture.py | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/testing/gate/session_fixture.py b/testing/gate/session_fixture.py index 73d2e8fd..078428b1 100644 --- a/testing/gate/session_fixture.py +++ b/testing/gate/session_fixture.py @@ -180,6 +180,13 @@ def run_setup(drv: TestAutomationServer, username: str, password: str, raise SessionUnavailable(f"wizard: input_fedid_label would not accept input ({e})") from e time.sleep(2.0) _reach(drv, "age_band_adult", lambda: drv.click("age_band_adult")) + # The legs run against a bare node with no LLM, so answer "run without AI": + # it removes the AI step, whose Next waits for a usable LLM choice + # (SetupState: AI -> hasUsableLlmChoice). Desktop already defaults there; + # iOS asks, and the walk stopped at 'ai' with Next disabled. + if "opt_run_without_ai" in _tags(drv): + _reach(drv, "opt_run_without_ai", lambda: drv.click("opt_run_without_ai")) + time.sleep(1.0) # Advance until the claim takes over. Bounded: a wizard that stops advancing # must say so rather than spin. From 3f5a6f5952bf2a9e225e2d6a77fafbf8fbc5ac42 Mon Sep 17 00:00:00 2001 From: Eric Moore Date: Mon, 28 Sep 2026 21:43:43 -0500 Subject: [PATCH 16/19] fix(gate): on a compact layout, open the one card a tab lists instead of stopping on CircleTab nav_map drops the row hop for a one-card tab because the wide layout opens it directly; phones list it first, so iOS landed on CircleTab. Open the only nav_epistemic_* row; with several, name them instead of guessing. Test added. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_0155SkTGdbwnSnR6tWBqJvUM --- testing/gate/run_flows.py | 14 ++++++++++++++ testing/test_flows.py | 26 ++++++++++++++++++++++++++ 2 files changed, 40 insertions(+) diff --git a/testing/gate/run_flows.py b/testing/gate/run_flows.py index ac91d795..fa5e29ee 100644 --- a/testing/gate/run_flows.py +++ b/testing/gate/run_flows.py @@ -133,6 +133,20 @@ async def navigate(helper, screen: str, chain: Sequence[str], *, hop_timeout: fl if not await helper.click(tag, timeout=int(hop_timeout * 1000)): return f"navigation to Screen.{screen}: clicking hop tag {tag!r} failed, {where}" got = await _settle_on(helper, screen, arrive_timeout) + if got == "CircleTab" and screen != "CircleTab": + # A tab with ONE card opens it directly in the wide layout (nav_map + # drops the row hop), but the compact layout (phones) lists the one + # card first. Open it when there is exactly one row; otherwise name + # the rows rather than guess. + rows = sorted({e.test_tag for e in await helper.get_elements() + if e.test_tag.startswith("nav_epistemic_")}) + if len(rows) == 1: + await helper.click(rows[0], timeout=int(hop_timeout * 1000)) + got = await _settle_on(helper, screen, arrive_timeout) + elif rows: + return (f"navigation to Screen.{screen}: walked {' -> '.join(chain)} and landed on " + f"the tab's card list with {len(rows)} rows ({', '.join(rows)}); nav_map " + f"expected one card") if got != screen: return (f"navigation to Screen.{screen}: walked {' -> '.join(chain)} and landed on " f"{got!r}") diff --git a/testing/test_flows.py b/testing/test_flows.py index 5ea635c8..67a382f4 100644 --- a/testing/test_flows.py +++ b/testing/test_flows.py @@ -517,3 +517,29 @@ def test_the_real_nav_map_reaches_the_seeded_flows_first_screens(): for spec in run_flows.load_flows([FLOWS]): start = spec.steps[0].requires.screen assert start in hops or start in flow_only, f"{spec.flow}: no way to Screen.{start}" + + +def test_navigate_opens_the_single_card_when_a_compact_tab_lists_it(): + """Phones list a one-card tab before opening it (iOS, #97); the runner + opens the only row instead of reporting 'landed on CircleTab'.""" + import asyncio + from testing.gate import run_flows + + class E: + def __init__(self, t): self.test_tag = t + + class H: + def __init__(self): self.screen = "Login"; self.clicked = [] + async def wait_for_element(self, tag, timeout=0): return True + async def click(self, tag, timeout=0): + self.clicked.append(tag) + self.screen = {"tab_people": "CircleTab", "nav_epistemic_contacts": "Contacts"}.get(tag, self.screen) + return True + async def get_screen(self): return self.screen + async def get_elements(self): return [E("nav_epistemic_contacts"), E("tab_people")] + + h = H() + got = asyncio.run(run_flows.navigate(h, "Contacts", ["circle_agent", "tab_people"], + hop_timeout=0.1, arrive_timeout=0.1)) + assert got is None, got + assert h.clicked[-1] == "nav_epistemic_contacts" From 2a83006e0edcb8d9c5f977eb3c9517d4f8d77e32 Mon Sep 17 00:00:00 2001 From: Eric Moore Date: Mon, 28 Sep 2026 22:30:55 -0500 Subject: [PATCH 17/19] fix(gate): on a multi-card tab list, open the row for the target screen iOS landed on a People list with the community roster and Contacts: the compact layout was still on Neighbours after the circle hop. Pick the row named for the target screen when the list has it; otherwise say which rows were there and ask whether the circle hop applied. Test added. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_0155SkTGdbwnSnR6tWBqJvUM --- testing/gate/run_flows.py | 11 ++++++++--- testing/test_flows.py | 26 ++++++++++++++++++++++++++ 2 files changed, 34 insertions(+), 3 deletions(-) diff --git a/testing/gate/run_flows.py b/testing/gate/run_flows.py index fa5e29ee..bc5010b4 100644 --- a/testing/gate/run_flows.py +++ b/testing/gate/run_flows.py @@ -44,6 +44,7 @@ import argparse import asyncio +import re import json import sys import time @@ -140,13 +141,17 @@ async def navigate(helper, screen: str, chain: Sequence[str], *, hop_timeout: fl # the rows rather than guess. rows = sorted({e.test_tag for e in await helper.get_elements() if e.test_tag.startswith("nav_epistemic_")}) - if len(rows) == 1: - await helper.click(rows[0], timeout=int(hop_timeout * 1000)) + # The row for THIS screen, when the list has it (Contacts sits in every + # circle's People tab; a tab can list several cards). + want = "nav_epistemic_" + re.sub(r"(? '.join(chain)} and landed on " f"the tab's card list with {len(rows)} rows ({', '.join(rows)}); nav_map " - f"expected one card") + f"expected one card, and none is {want!r} — was the circle hop applied?") if got != screen: return (f"navigation to Screen.{screen}: walked {' -> '.join(chain)} and landed on " f"{got!r}") diff --git a/testing/test_flows.py b/testing/test_flows.py index 67a382f4..a8d23e12 100644 --- a/testing/test_flows.py +++ b/testing/test_flows.py @@ -543,3 +543,29 @@ async def get_elements(self): return [E("nav_epistemic_contacts"), E("tab_people hop_timeout=0.1, arrive_timeout=0.1)) assert got is None, got assert h.clicked[-1] == "nav_epistemic_contacts" + + +def test_navigate_prefers_the_target_row_when_a_tab_lists_several(): + import asyncio + from testing.gate import run_flows + + class E: + def __init__(self, t): self.test_tag = t + + class H: + def __init__(self): self.screen = "Login"; self.clicked = [] + async def wait_for_element(self, tag, timeout=0): return True + async def click(self, tag, timeout=0): + self.clicked.append(tag) + self.screen = {"tab_people": "CircleTab", "nav_epistemic_contacts": "Contacts", + "nav_epistemic_community_roster": "CommunityRoster"}.get(tag, self.screen) + return True + async def get_screen(self): return self.screen + async def get_elements(self): + return [E("nav_epistemic_community_roster"), E("nav_epistemic_contacts")] + + h = H() + got = asyncio.run(run_flows.navigate(h, "Contacts", ["circle_agent", "tab_people"], + hop_timeout=0.1, arrive_timeout=0.1)) + assert got is None, got + assert h.clicked[-1] == "nav_epistemic_contacts" From 917a9a1b0545e56c934df407b115b2e3c4988c03 Mon Sep 17 00:00:00 2001 From: Eric Moore Date: Mon, 28 Sep 2026 23:05:34 -0500 Subject: [PATCH 18/19] chore: re-record the vendoring digest and route baseline after merging main Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_0155SkTGdbwnSnR6tWBqJvUM --- client/VENDORING.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/client/VENDORING.md b/client/VENDORING.md index 8696269d..08a3dc81 100644 --- a/client/VENDORING.md +++ b/client/VENDORING.md @@ -40,7 +40,7 @@ source is the pair a bisect wants: The tree's current recorded state — sha256-of-sha256s over every git-tracked file under `client/` except this one: -**state digest:** `dece5915e6100c07639cb5cb597ee35cd9701d527ca0588969aa5e7ceba3b2fc` +**state digest:** `1ab9a22b7e23dea8b1e2dc3aac8a8c6e3cc95c5ca7d6783a4d73bf40321f2bbd` `packaging/check_vendoring.py` asserts it on every push, and refuses any tracked file matching a §2 never-vendor class. **Any commit that touches From b1c7199e7e7cadf9965e02ae1af167c84c275f81 Mon Sep 17 00:00:00 2001 From: Eric Moore Date: Mon, 28 Sep 2026 23:07:47 -0500 Subject: [PATCH 19/19] test(flows): prove the proposed-tag refusal with a tag a real CSD still marks proposed CSD-005's trust chip shipped on main (people review), so the test that named it went red for the product doing its job. Derive the tag instead (the same fix the two-node branch made). Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_0155SkTGdbwnSnR6tWBqJvUM --- testing/test_flows.py | 38 +++++++++++++++++++++++++++++++------- 1 file changed, 31 insertions(+), 7 deletions(-) diff --git a/testing/test_flows.py b/testing/test_flows.py index a8d23e12..ee31e20d 100644 --- a/testing/test_flows.py +++ b/testing/test_flows.py @@ -153,19 +153,43 @@ def test_a_proposed_tag_named_anywhere_in_a_flow_is_a_load_error(tmp_path, where FlowSpec.load(_flow(tmp_path, body), csd_root=_csd_root(tmp_path)) -def test_the_real_csd_005_refuses_its_proposed_trust_chip(tmp_path): - """Against the shipped document, not a fixture: `contacts_row_trust` is - `proposed:` in CSD-005, so no flow may assert it yet.""" - body = """\ +def _a_real_proposed_tag(): + """(CSD id, a tag it still marks `proposed:`) from the shipped documents. + + Derived, not named: this test used CSD-005's `contacts_row_trust`, and the + chip shipped — the document stopped marking it proposed and the test went + red for the product doing its job. Any real CSD with a proposed tag proves + the same thing.""" + import re as _re # noqa: PLC0415 + from testing.gate.csd_doc import CsdError, load # noqa: PLC0415 + for path in sorted((REPO / "FSD" / "CSD").glob("CSD-*.md")): + m = _re.match(r"(CSD-\d+)", path.name) + try: + doc = load(m.group(1)) if m else None + except CsdError: + continue + if doc is not None and doc.proposed: + return doc.csd_id, sorted(doc.proposed)[0] + return None + + +def test_a_real_csd_refuses_a_tag_it_still_marks_proposed(tmp_path): + """Against a shipped document, not a fixture: a tag a real CSD still marks + `proposed:` may not be asserted by a flow.""" + found = _a_real_proposed_tag() + if found is None: + pytest.skip("no shipped CSD marks any tag proposed") + csd_id, tag = found + body = f"""\ flow: p - csd: CSD-005 + csd: {csd_id} steps: - step_id: s title: t expect: - visible: [contacts_row_trust] + visible: [{tag}] """ - with pytest.raises(SpecError, match="contacts_row_trust.*proposed"): + with pytest.raises(SpecError, match=f"{tag}.*proposed"): FlowSpec.load(_flow(tmp_path, body))