diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index c584bb7c..ab085428 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -111,7 +111,7 @@ jobs: python3 -m pip install --quiet pytest pyyaml python3 -m pytest testing/test_driver_rules.py testing/test_gate_vendoring.py \ testing/test_bringup.py testing/test_five_platform_workflow.py \ - testing/test_csd_state_tags.py -q + testing/test_flows.py testing/test_csd_state_tags.py -q - name: A published version must offer a universal wheel # Only for versions already on the index — a PR's VERSION is normally diff --git a/.github/workflows/five-platform-live-qa.yml b/.github/workflows/five-platform-live-qa.yml index bb85b291..fb4f218c 100644 --- a/.github/workflows/five-platform-live-qa.yml +++ b/.github/workflows/five-platform-live-qa.yml @@ -141,6 +141,12 @@ jobs: with: python-version: '3.10' + # THE CSD FLOWS NEED PyYAML (testing/gate/flow_spec.py reads the flow and + # its CSD's typed blocks). Everything else the legs run is stdlib; this is + # the one install, and it goes into the interpreter every leg below uses. + - name: The flow runner's one dependency + run: python3 -m pip install --quiet pyyaml + - name: The node this client will be a client of id: node env: @@ -224,12 +230,18 @@ jobs: - name: Start the node uses: ./.github/actions/ciris-node + # THE SMOKE WALK, THEN THE CSD FLOWS, IN ONE PROCESS AGAINST ONE APP. + # `--flows` runs every testing/flows/*.yaml after the walk passes, in the + # app the walk just proved, signing in once (session_fixture). A flow + # refused by its `client:` floor is reported and does not redden the leg; + # a flow that fails, or never reaches its first screen, does. See + # testing/flows/README.md. Every leg below passes the same flag. - name: Linux desktop run: | python3 -m testing.gate.run_platform --platform desktop --xvfb \ --jar "${{ steps.art.outputs.jar }}" \ --node-version "${{ steps.node.outputs.version }}" \ - --shots shots --report reports/linux.json + --shots shots --report reports/linux.json --flows testing/flows # WITHOUT THIS THE EMULATOR RUNS IN SOFTWARE AND DIES. # @@ -289,7 +301,7 @@ jobs: # # after the emulator had booted and the whole leg had been paid for. # It reads worse on one line and it is the form that runs. - script: python3 -m testing.gate.run_platform --platform android --apk "${{ steps.art.outputs.apk }}" --node-version "${{ steps.node.outputs.version }}" --shots shots --report reports/android.json; rc=$?; adb logcat -d > logcat.txt 2>&1; exit $rc + script: python3 -m testing.gate.run_platform --platform android --apk "${{ steps.art.outputs.apk }}" --node-version "${{ steps.node.outputs.version }}" --shots shots --report reports/android.json --flows testing/flows; rc=$?; adb logcat -d > logcat.txt 2>&1; exit $rc # ALWAYS. A failure you cannot diagnose from the artifact costs a re-run # to learn what this run already knew. @@ -325,6 +337,17 @@ jobs: distribution: temurin java-version: '17' + # A PROVISIONED INTERPRETER for the macOS desktop leg, so the flow runner's + # PyYAML goes into a Python this job owns rather than the image's + # externally-managed one. The iOS leg switches to 3.10 further down and + # installs it again there. + - uses: actions/setup-python@v5 + with: + python-version: '3.12' + + - name: The flow runner's one dependency + run: python3 -m pip install --quiet pyyaml + - name: The node this client will be a client of id: node env: @@ -350,7 +373,7 @@ jobs: python3 -m testing.gate.run_platform --platform desktop \ --jar "$(python3 -m testing.gate.candidate_artifacts --kind desktop | tail -1)" \ --node-version "${{ steps.node.outputs.version }}" \ - --shots shots --report reports/macos.json + --shots shots --report reports/macos.json --flows testing/flows # ── THE iOS BUNDLE, MATERIALIZED THE WAY THE AGENT'S GATE DOES IT ────── # @@ -541,6 +564,9 @@ jobs: - name: iOS simulator run: | set -euo pipefail + # `python3` is the 3.10 set up above now, not the 3.12 the macOS leg + # installed PyYAML into — the flow runner needs it here too. + python3 -m pip install --quiet pyyaml # THE TASK NAME, AND THE DIRECTORY, BOTH WRONG — AND THE SECOND HID THE FIRST. # # `:shared:assembleDebugXCFramework` does not exist. The framework is @@ -632,7 +658,7 @@ jobs: rc=0 python3 -m testing.gate.run_platform --platform ios --app "$app" --udid "$UDID" \ --node-version "${{ steps.node.outputs.version }}" \ - --shots shots --report reports/ios.json || rc=$? + --shots shots --report reports/ios.json --flows testing/flows || rc=$? # THE APP'S OWN ACCOUNT, EITHER WAY. Run 35359571538 got the iOS app # to a real screen — "Engine Failed to Start: server did not become @@ -732,10 +758,11 @@ jobs: - name: Windows desktop shell: bash run: | + python3 -m pip install --quiet pyyaml python3 -m testing.gate.run_platform --platform desktop \ --jar "$(python3 -m testing.gate.candidate_artifacts --kind desktop | tail -1)" \ --node-version "${{ steps.node.outputs.version }}" \ - --shots shots --report reports/windows.json + --shots shots --report reports/windows.json --flows testing/flows - if: always() uses: actions/upload-artifact@v4 diff --git a/client/VENDORING.md b/client/VENDORING.md index 8696269d..08a3dc81 100644 --- a/client/VENDORING.md +++ b/client/VENDORING.md @@ -40,7 +40,7 @@ source is the pair a bisect wants: The tree's current recorded state — sha256-of-sha256s over every git-tracked file under `client/` except this one: -**state digest:** `dece5915e6100c07639cb5cb597ee35cd9701d527ca0588969aa5e7ceba3b2fc` +**state digest:** `1ab9a22b7e23dea8b1e2dc3aac8a8c6e3cc95c5ca7d6783a4d73bf40321f2bbd` `packaging/check_vendoring.py` asserts it on every push, and refuses any tracked file matching a §2 never-vendor class. **Any commit that touches diff --git a/client/shared/src/androidMain/kotlin/ai/ciris/mobile/shared/testing/TestAutomationServer.android.kt b/client/shared/src/androidMain/kotlin/ai/ciris/mobile/shared/testing/TestAutomationServer.android.kt index 30497731..3d131da8 100644 --- a/client/shared/src/androidMain/kotlin/ai/ciris/mobile/shared/testing/TestAutomationServer.android.kt +++ b/client/shared/src/androidMain/kotlin/ai/ciris/mobile/shared/testing/TestAutomationServer.android.kt @@ -94,7 +94,9 @@ class AndroidTestAutomationServer(private val port: Int = 9091) { // Input text to element post("/input") { val request = call.receive() - call.respond(TestAutomationHandler.handleInput(request)) + val resp = TestAutomationHandler.handleInput(request) + // A failed input is not a 200 (desktop answers 404; iOS now does too). + call.respond(if (resp.success) HttpStatusCode.OK else HttpStatusCode.NotFound, resp) } // Scroll the screen (recovery after an off-screen refusal) diff --git a/client/shared/src/iosMain/kotlin/ai/ciris/mobile/shared/testing/TestAutomationServer.ios.kt b/client/shared/src/iosMain/kotlin/ai/ciris/mobile/shared/testing/TestAutomationServer.ios.kt index 6cc5540f..e166d3bc 100644 --- a/client/shared/src/iosMain/kotlin/ai/ciris/mobile/shared/testing/TestAutomationServer.ios.kt +++ b/client/shared/src/iosMain/kotlin/ai/ciris/mobile/shared/testing/TestAutomationServer.ios.kt @@ -265,7 +265,11 @@ class IOSTestAutomationServer(private val port: Int = 9091) { } method == "POST" && path == "/input" -> { val req = json.decodeFromString(body) - 200 to json.encodeToString(TestAutomationHandler.handleInput(req)) + val resp = TestAutomationHandler.handleInput(req) + // A failed input is not a 200: desktop answers 404, and a + // harness that checks the status (the gate's driver did) + // otherwise believes it typed into a field that took nothing. + (if (resp.success) 200 else 404) to json.encodeToString(resp) } method == "POST" && path == "/wait" -> { val req = json.decodeFromString(body) diff --git a/testing/driver.py b/testing/driver.py index 2d383f77..02b4e184 100644 --- a/testing/driver.py +++ b/testing/driver.py @@ -138,9 +138,15 @@ def _call(self, method: str, route: str, body: dict | None = None) -> Any: if not raw.strip(): return None try: - return json.loads(raw) + parsed = json.loads(raw) except json.JSONDecodeError: return raw + # A body that says it failed has failed, whatever the HTTP status: the + # iOS server answered a failed /input with 200 until 0.5.225, and the + # walk then "typed" into fields that took nothing. + if isinstance(parsed, dict) and parsed.get("success") is False: + raise DriverError(f"{method} {route} -> {parsed.get('error') or parsed}") + return parsed # ---- reads -------------------------------------------------------- diff --git a/testing/flows/README.md b/testing/flows/README.md new file mode 100644 index 00000000..7b97996b --- /dev/null +++ b/testing/flows/README.md @@ -0,0 +1,142 @@ +# CSD flows + +A CSD's §4 says what its surface must do. A flow here is that §4 made +executable, and the five-platform gate (`five-platform-live-qa.yml`) runs every +flow in this directory on every leg — Linux, macOS and Windows desktops, the +Android emulator, the iOS simulator — against a real `ciris-server`. + +That is what `testable` means in `CSD.md` §1: *"floor flips off unreleased; flow +runs on the matrix"*. + +## The file + +```yaml +flow: people # the flow's id; unique across this directory +csd: CSD-005 # the CSD it tests — REQUIRED here +title: People on a node with no contacts yet +description: >- # optional; say what is NOT driven and why + … +client: ">=0.5.224" # the client that carries every tag it names + +steps: + - step_id: landing + title: Signing in on a bare node lands on People + requires: # checked BEFORE the step; the first step's is the entry + screen: Contacts + do: # click / input / scroll_to / wait (one per entry) + - wait: card_contacts_add + wait_ms: 5000 # a `wait` waits wait_ms × 4 + expect: # checked AFTER + visible: [card_contacts_add] + absent: [contacts_empty] + + - step_id: search_no_match + title: A search nothing matches is the empty state + do: + - input: {input_contacts_search: "zz-no-such-contact"} + expect: + state: empty # read from the CSD's `states:` block +``` + +The language is `testing/gate/flow_spec.py`'s — vendored from CIRISAgent, with +the CSD/3 §3 predicates (`count`, `number`, `matches`, `one_of`, `each`, +`relation`, `state`) and one local addition, `csd:` (see +`testing/gate/VENDORED.md`). Unknown keys anywhere are a load error. + +### How a flow is tied to its CSD + +`csd: CSD-NNN` resolves to the one `FSD/CSD/CSD-NNN-*.md`, and the runner reads +two of its typed blocks (`testing/gate/csd_doc.py`): + +- **`csd:shows`** → field id → tag. A `relation:` names `ceg:` field ids + (`capacity:composite`), not tags, and this is how they resolve. +- **`csd:states`** → state → tag. `state: empty` holds when the `empty` tag is on + screen **and no other state's tag is** — so an error cannot pass for "nothing + here". + +Each of these is a **load error**, found before any app is started: + +- the CSD does not exist, is ambiguous, or a typed block does not parse; +- the flow names a tag the CSD still marks `proposed:` — in `requires`, + `expect` or `do`. No client carries it, and it would fail as "element not + found", which looks exactly like a broken app; +- `state: X` where the CSD gives X no tag, or a proposed one; +- a `relation:` operand that is not one of the CSD's `shows:` fields. + +`testing/test_flows.py` also checks that every literal tag a flow here names is +a string in the client's `commonMain` source. + +## Verdicts + +| verdict | when | leg | +|---|---|---| +| `pass` | every step held | green | +| `refused` | the `client:` floor is above the client under test (`>=X`, `>X`, `unreleased`) | green — reported, not passed | +| `cannot-start` | floor met, but the flow never reached its first screen (no hop, a hop tag missing, or the first `requires` never held) | **red** | +| `fail` | it started and a step broke | **red** | + +`cannot-start` is red on purpose. The floor is how a flow waits for a surface +that has not shipped; once the floor is met, a flow that never reached its first +screen is a flow that silently never ran. + +Each leg's report (`reports/.json`) carries a `flows` list with every +outcome and its step-level detail; screenshots and per-flow JSON land under +`shots/flows-/`. The client version the floor is checked against is this +tree's `VERSION` — on the matrix the artifact is asserted to be this tree's. + +## From `building` to `testable` + +1. Every tag the flow needs is real: no `proposed:` left on the rows it drives, + and the PR that adds them has shipped. +2. Write `testing/flows/.yaml` with `csd:` pointing at the CSD and + `client: ">="`. Until a release carries it, use + `client: unreleased` — the flow loads, is checked, and is refused on the + matrix instead of failing. +3. Run it locally (below), then let the nightly matrix run it. +4. When it is **green on the platforms the CSD's §5 declares**, the pen-holder + moves `stage:` to `testable`. Nothing advances the stage automatically — a + green run is evidence for the edit, not the edit. + +## Running one flow locally, against a desktop client + +The runner drives whatever client answers on the test-automation port. Keep it +off your own install: the node takes `--home`, the client reads `CIRIS_HOME` +(`testing/gate/session_fixture.py` explains why they differ). + +```bash +# 1. a throwaway node +python3 -m testing.gate.node_fixture --version v0.5.224 --home /tmp/flows-node \ + --platform x86_64-unknown-linux-gnu # or aarch64-apple-darwin + +# 2. the desktop client in test mode, with its own home +( cd client && ./gradlew :desktopApp:packageUberJarForCurrentOS ) +CIRIS_TEST_MODE=true CIRIS_HOME=/tmp/flows-client \ + java -jar "$(python3 -m testing.gate.candidate_artifacts --kind desktop | tail -1)" & + +# 3. the flow — it signs in (running first-run setup if the node has no owner) +python3 -m testing.gate.run_flows --platform desktop \ + --flows testing/flows/people.yaml --report /tmp/flows.json +``` + +`--client-version` checks floors against another version (default: `VERSION`); +`--no-sign-in` drives whatever screen the client is already on; `--url` points at +a test server other than `http://127.0.0.1:9091`. Exit status is 0 only when +every flow passed or was refused. + +On the matrix the same code runs inside `run_platform` (`--flows testing/flows`) +after the smoke walk, in the app the walk just brought up, so there is one +bring-up per leg and one session. + +## What this does not do yet + +- **Navigation is to the first screen only.** Before step one the runner walks + the hop `testing/gate/nav_map.py` derives for the flow's first + `requires: screen:` on this build (node or agent tree, from `/state`), + waiting for each tag before clicking it. A missing hop tag is `cannot-start` + naming the tag; a screen with no hop that is not flow-only is `cannot-start` + with "no nav hop for Screen.X"; a flow-only screen (pre-login, wizards, + leaves) is waited for, not walked to. Hops between later steps are the flow's + own `do:` clicks. +- **Only what a bare node can show.** The matrix stands up one node with no + contacts, no agent and no peers, so CSD-005's populated list and receipt sheet + are not driven here. diff --git a/testing/flows/people.yaml b/testing/flows/people.yaml new file mode 100644 index 00000000..7fb599f4 --- /dev/null +++ b/testing/flows/people.yaml @@ -0,0 +1,62 @@ +flow: people +csd: CSD-005 +title: People on a node with no contacts yet +description: >- + The matrix's first CSD flow, and the one that proves the wiring: every leg signs + in on a bare ciris-server, which has no contacts, so People lands in the + "empty and unsearched" shape CSD-005 §4 describes — the add card INSTEAD of an + empty block. A search nothing matches then shows the empty state proper, and + clearing it brings the add card back. Every tag here ships in 0.5.224 + (PeopleTags in ContactsScreen/PeopleSupport.kt). + + Not driven here, and why: the populated list and the receipt sheet need a + second node to be a contact of, which the matrix does not stand up; the + `proposed:` trust chip cannot be named by a flow at all. +client: ">=0.5.224" + +steps: + - step_id: landing + title: Signing in on a bare node lands on People, offering to add someone + description: >- + `requires` is the entry precondition, stated rather than assumed: a client + that landed anywhere else reports "cannot start", not "a People element is + broken". The wait absorbs the first /v1/contacts read (the loading state). + requires: + screen: Contacts + do: + - wait: card_contacts_add + wait_ms: 5000 + expect: + screen: Contacts + visible: [input_contacts_search, card_contacts_add, input_contacts_add_key, + btn_contacts_add_submit, btn_contacts_refresh] + # The add card stands in for the empty block over an EMPTY, UNSEARCHED + # list; and a node that answered is neither loading, nor in error, nor + # too old for contacts. + absent: [contacts_empty, contacts_list, contacts_loading, contacts_error, + contacts_unsupported] + + - step_id: search_no_match + title: A search nothing matches is the empty state, not an error + description: >- + `state: empty` is checked against CSD-005's `states:` block: contacts_empty + on screen, and the populated, loading and error tags all absent — so an + error cannot pass for "no matches". + do: + - input: {input_contacts_search: "zz-no-such-contact"} + - wait: contacts_empty + wait_ms: 2500 + expect: + state: empty + absent: [card_contacts_add] + + - step_id: clear_search + title: Clearing the search brings the add card back + do: + - input: {input_contacts_search: ""} + - wait: card_contacts_add + wait_ms: 2500 + expect: + screen: Contacts + visible: [card_contacts_add] + absent: [contacts_empty, contacts_error] diff --git a/testing/gate/VENDORED.md b/testing/gate/VENDORED.md index 122eb552..a4b58050 100644 --- a/testing/gate/VENDORED.md +++ b/testing/gate/VENDORED.md @@ -67,6 +67,24 @@ tagged is drivable" about a screen with no elements on it. `check_csd.py`, and two implementations of one DSL is the drift this repo exists to measure. +### `flow_spec.py` — local delta: a flow names its CSD (`csd:`) + +Added so CSD flows can run on this repo's matrix (`testing/flows/`, +`testing/gate/run_flows.py`). Each change is marked `LOCAL DELTA` in the source: + +| change | reason | +|---|---| +| `csd` added to `_FLOW_KEYS`; `FlowSpec.csd_id` / `FlowSpec.csd` | a flow says which CSD it tests, so the runner can read that CSD's `shows:` (→ `field_tags`, for `relation`) and `states:` (→ `state_tags`, for `state:`) instead of every caller wiring them by hand | +| `FlowSpec.load(path, csd_root=None)` binds the CSD at load | a CSD that is missing, ambiguous or does not parse is a **load error**, never a skip — a flow that silently loses its CSD loses the checks that make `state:`/`relation:` mean anything. The CSD is read by `csd_doc.py` (ours), which takes its block grammar from `packaging/check_csd_v3.py` rather than re-typing it | +| a `proposed:` tag named in `requires`/`expect`/`do`, a `state:` whose tag is proposed or absent, or a `relation` operand that is not a `shows:` field → load error | no client carries a proposed tag, so asserting it fails as "element not found" — indistinguishable from a broken app. `count`/`each` globs are deliberately NOT checked: CSD-005's real `contacts_row_*` rows share a prefix with its proposed `contacts_row_trust` chip | +| `state:` is now CHECKED: that state's tag on screen, every other state's tag not; with no `states:` map it fails | upstream's `state:` body was `pass` — the one predicate CSD/3 makes mandatory asserted nothing, a vacuous green. Upstream flows do not use `state:`, so none of them changes behaviour | +| `FlowRunner(state_tags=…)`; `run()` fills both maps from `spec.csd` when the caller did not | one source for the maps: the CSD | +| `write_report` records `csd` | a result that cannot say what it tested cannot be acted on | + +A flow without `csd:` still loads and runs exactly as before; `run_flows.py` +is what requires the key for flows in this repo. Upstream needs the same key +before CIRISAgent's flows can be checked against their CSDs the same way. + ### `platforms.py` — and a mistake this file previously recorded as a fact An earlier version of this document said all three vendored modules were diff --git a/testing/gate/csd_doc.py b/testing/gate/csd_doc.py new file mode 100644 index 00000000..0ec74cde --- /dev/null +++ b/testing/gate/csd_doc.py @@ -0,0 +1,151 @@ +"""The half of a CSD a flow needs: its `shows:` fields and its `states:` tags. + +A flow in `testing/flows/` names its CSD (`csd: CSD-005`). This module finds that +document and reads the two typed blocks the runner cannot work without: + + * `csd:shows` -> `field_tags`, the `ceg:` field id -> the tag drawing it. A + `relation:` operand is a field id, so without this a flow can only relate + boxes rather than constitutional values (CSD/3 §3). + * `csd:states` -> `state_tags`, state -> tag. `state: empty` is asserted as + "that state's tag is on screen, and every other state's tag is not" — so an + error cannot pass for an empty list, which is the one confusion CSD/3 §2.2 + makes mandatory to prevent. + +It also records which tags are still `proposed:`, because a flow may not drive or +assert a tag no client has shipped: that would fail as "element not found", which +is indistinguishable from a broken app. + +ONE BLOCK GRAMMAR. The fenced-block pattern is `packaging/check_csd_v3.py`'s own +`BLOCK`, loaded from that file rather than re-typed here, so the checker and the +runner cannot disagree about what counts as a typed block. (It is loaded by path: +`import packaging` would find the PyPI package of that name first.) + +EVERY FAILURE IS A LOAD ERROR. A CSD that is missing, ambiguous or does not parse +raises `CsdError` — never a silent skip, because a flow that quietly loses its CSD +loses the checks that make its `state:` and `relation:` mean anything. +""" + +from __future__ import annotations + +import importlib.util +import re +from dataclasses import dataclass, field +from pathlib import Path +from typing import Dict, Optional, Set + +import yaml + +#: Where this repo keeps its CSDs. +DEFAULT_CSD_ROOT = Path(__file__).resolve().parents[2] / "FSD" / "CSD" +_CHECKER = Path(__file__).resolve().parents[2] / "packaging" / "check_csd_v3.py" +_ID = re.compile(r"^CSD-\d{3}$") +PROPOSED = "proposed:" + + +class CsdError(Exception): + """The CSD a flow names cannot be used. Always a load error.""" + + +def _block_pattern() -> "re.Pattern[str]": + spec = importlib.util.spec_from_file_location("_check_csd_v3", _CHECKER) + if spec is None or spec.loader is None: # pragma: no cover — the file is in the tree + raise CsdError(f"cannot load the CSD block grammar from {_CHECKER}") + mod = importlib.util.module_from_spec(spec) + spec.loader.exec_module(mod) + return mod.BLOCK + + +BLOCK = _block_pattern() + + +@dataclass +class CsdDoc: + csd_id: str + path: Path + stage: Optional[str] + #: `ceg:` field id -> tag. A parameterised id is keyed both as written + #: (`consent:{kind}`) and as bound (`consent:replication`). + field_tags: Dict[str, str] = field(default_factory=dict) + #: state name -> tag, for the states that name one. + state_tags: Dict[str, str] = field(default_factory=dict) + #: Every tag the CSD marks `proposed:` (prefix stripped). + proposed: Set[str] = field(default_factory=set) + #: Field ids whose tag is still proposed. + proposed_fields: Set[str] = field(default_factory=set) + #: State names whose tag is still proposed. + proposed_states: Set[str] = field(default_factory=set) + + +def _bound(ceg: str, bind: Dict[str, str]) -> str: + out = ceg + for k, v in bind.items(): + out = out.replace("{" + str(k) + "}", str(v)) + return out + + +def parse(path: Path, csd_id: str = "") -> CsdDoc: + """Read one CSD's typed blocks. Raises CsdError on anything unusable.""" + try: + text = path.read_text(encoding="utf-8") + except OSError as e: + raise CsdError(f"{path}: cannot be read ({e})") from e + + blocks: Dict[str, object] = {} + for name, body in BLOCK.findall(text): + try: + blocks[name] = yaml.safe_load(body) + except yaml.YAMLError as e: + raise CsdError(f"{path}: its `csd:{name}` block does not parse: {e}") from e + if "stage" not in blocks: + # Every CSD/3 document has one; a file without it is not a CSD, and a + # flow bound to it would be bound to prose. + raise CsdError(f"{path}: no `yaml csd:stage` block — not a CSD/3 document") + + doc = CsdDoc(csd_id=csd_id or path.stem, path=path, + stage=(blocks.get("stage") or {}).get("stage")) + + shows = blocks.get("shows") or {} + if not isinstance(shows, dict): + raise CsdError(f"{path}: `csd:shows` is not a mapping") + for i, f in enumerate(shows.get("fields") or []): + if not isinstance(f, dict) or not f.get("ceg"): + raise CsdError(f"{path}: csd:shows field[{i}] has no `ceg:`") + raw_tag = str(f.get("tag") or "") + if not raw_tag: + continue + proposed = raw_tag.startswith(PROPOSED) + tag = raw_tag[len(PROPOSED):] if proposed else raw_tag + ids = {str(f["ceg"]), _bound(str(f["ceg"]), f.get("bind") or {})} + for fid in ids: + doc.field_tags[fid] = tag + if proposed: + doc.proposed_fields.add(fid) + if proposed: + doc.proposed.add(tag) + + states = blocks.get("states") or {} + if not isinstance(states, dict): + raise CsdError(f"{path}: `csd:states` is not a mapping") + for name, row in states.items(): + raw_tag = str((row or {}).get("tag") or "") if isinstance(row, dict) else "" + if not raw_tag: + continue + if raw_tag.startswith(PROPOSED): + doc.proposed.add(raw_tag[len(PROPOSED):]) + doc.proposed_states.add(str(name)) + continue # a proposed state tag cannot be asserted; see flow_spec + doc.state_tags[str(name)] = raw_tag + return doc + + +def load(csd_id: str, root: Optional[Path] = None) -> CsdDoc: + """Find `CSD-NNN-*.md` under `root` and parse it.""" + if not isinstance(csd_id, str) or not _ID.match(csd_id): + raise CsdError(f"`csd: {csd_id!r}` is not a CSD id; write it as e.g. `CSD-005`") + root = Path(root) if root else DEFAULT_CSD_ROOT + hits = sorted(root.glob(f"{csd_id}-*.md")) + if not hits: + raise CsdError(f"{csd_id}: no {csd_id}-*.md under {root} — a flow cannot name a CSD that does not exist") + if len(hits) > 1: + raise CsdError(f"{csd_id}: ambiguous, {len(hits)} documents match: {[h.name for h in hits]}") + return parse(hits[0], csd_id) diff --git a/testing/gate/flow_spec.py b/testing/gate/flow_spec.py index 038c4f89..90e06f77 100644 --- a/testing/gate/flow_spec.py +++ b/testing/gate/flow_spec.py @@ -59,7 +59,9 @@ "count", "number", "matches", "one_of", "each", "relation", "state", } _ACTION_KEYS = {"click", "input", "scroll_to", "wait", "wait_ms"} -_FLOW_KEYS = {"flow", "title", "description", "client", "steps"} +#: LOCAL DELTA (VENDORED.md): `csd` names the CSD a flow tests, so the runner can +#: read that CSD's `shows:` (for `relation` field ids) and `states:` (for `state:`). +_FLOW_KEYS = {"flow", "title", "description", "client", "steps", "csd"} _RELATION_OPS = {"eq", "ne", "lt", "lte", "gt", "gte", "min_of", "max_of", "sum_of"} @@ -70,6 +72,14 @@ class SpecError(Exception): """The spec itself is wrong. Raised at load time, before anything is driven.""" +def _tags_of(cond: "Condition") -> List[str]: + """Every literal tag a condition names (globs and field ids excluded).""" + out = list(cond.visible) + list(cond.absent) + for d in (cond.text, cond.number, cond.matches, cond.one_of): + out.extend(d.keys()) + return out + + @dataclass class Condition: """A `requires` or `expect` block: what must be true at a point in the flow.""" @@ -251,9 +261,12 @@ class FlowSpec: description: str = "" client_floor: Optional[str] = None path: Optional[Path] = None + #: LOCAL DELTA: the CSD this flow tests, when it names one (`csd: CSD-005`). + csd_id: Optional[str] = None + csd: Any = None # testing.gate.csd_doc.CsdDoc @classmethod - def load(cls, path: Path) -> "FlowSpec": + def load(cls, path: Path, csd_root: Optional[Path] = None) -> "FlowSpec": try: raw = yaml.safe_load(path.read_text(encoding="utf-8")) except yaml.YAMLError as exc: @@ -274,7 +287,7 @@ def load(cls, path: Path) -> "FlowSpec": if s.step_id in seen: raise SpecError(f"{path}: duplicate step_id {s.step_id!r} (steps {seen[s.step_id]} and {i})") seen[s.step_id] = i - return cls( + spec = cls( flow=str(raw["flow"]), title=str(raw.get("title") or raw["flow"]), description=str(raw.get("description") or ""), @@ -282,6 +295,74 @@ def load(cls, path: Path) -> "FlowSpec": steps=steps, path=path, ) + if "csd" in raw: + spec._bind_csd(raw["csd"], csd_root) + return spec + + def _bind_csd(self, csd_id: Any, csd_root: Optional[Path]) -> None: + """LOCAL DELTA: load the named CSD and hold the flow to it. + + A CSD that is missing or does not parse is a LOAD ERROR, never a skip — + the flow would otherwise run with `state:` and `relation:` unanchored. + And a flow may not name a `proposed:` tag anywhere: no client carries + it, so driving or asserting it fails as "element not found", which is + indistinguishable from a broken app. + """ + from testing.gate.csd_doc import CsdError, load as load_csd # noqa: PLC0415 + + where = f"{self.path}" + try: + doc = load_csd(csd_id, csd_root) + except CsdError as e: + raise SpecError(f"{where}: `csd:` {e}") from e + self.csd_id, self.csd = str(csd_id), doc + + for step in self.steps: + at = f"{where}: step {step.step_id!r}" + for half, cond in (("requires", step.requires), ("expect", step.expect)): + for tag in _tags_of(cond): + if tag in doc.proposed: + raise SpecError( + f"{at}.{half} names {tag!r}, which {doc.csd_id} still marks " + f"`proposed:` — a flow cannot assert a tag no client carries" + ) + # Globs (`count`/`each` `of:`) are NOT checked against the + # proposed set: CSD-005's real `contacts_row_*` rows share a + # prefix with its proposed `contacts_row_trust` chip, and + # refusing the glob would refuse a flow that names no proposed + # tag. `each` over an empty match already fails at run time. + if cond.state is not None: + if cond.state in doc.proposed_states: + raise SpecError( + f"{at}.{half} asserts `state: {cond.state}`, whose tag " + f"{doc.csd_id} still marks `proposed:`" + ) + if cond.state not in doc.state_tags: + raise SpecError( + f"{at}.{half} asserts `state: {cond.state}`, but {doc.csd_id}'s " + f"`states:` names no tag for it, so it cannot be checked" + ) + if cond.relation is not None: + r = cond.relation + operands = [r["left"]] + list(r.get("of") or []) + ( + [r["right"]] if r.get("right") else []) + for fid in map(str, operands): + if fid in doc.proposed_fields: + raise SpecError( + f"{at}.{half} relates {fid!r}, whose tag {doc.csd_id} " + f"still marks `proposed:`" + ) + if ":" in fid and fid not in doc.field_tags: + raise SpecError( + f"{at}.{half} relates {fid!r}, which is not a field " + f"of {doc.csd_id}'s `shows:`" + ) + for action in step.do: + if action.target in doc.proposed: + raise SpecError( + f"{at}.do drives {action.target!r}, which {doc.csd_id} still " + f"marks `proposed:`" + ) def check_client_floor(floor: Optional[str], actual: Optional[str]) -> Optional[str]: @@ -354,7 +435,8 @@ class FlowRunner: """Executes a FlowSpec against a connected DesktopAppHelper.""" def __init__(self, helper, platform=None, artifacts: Optional[Path] = None, - field_tags: Optional[Dict[str, str]] = None) -> None: + field_tags: Optional[Dict[str, str]] = None, + state_tags: Optional[Dict[str, str]] = None) -> None: self.helper = helper self.platform = platform self.artifacts = Path(artifacts) if artifacts else None @@ -363,6 +445,9 @@ def __init__(self, helper, platform=None, artifacts: Optional[Path] = None, #: block. `relation` operands are field ids, so without this a flow #: could only relate boxes rather than constitutional values. self.field_tags: Dict[str, str] = dict(field_tags or {}) + #: LOCAL DELTA: state -> tag, from the CSD's `states:` block. Filled + #: from the spec's CSD at `run()` when the caller does not pass it. + self.state_tags: Dict[str, str] = dict(state_tags or {}) async def _drivable(self) -> List[str]: """Tags actually ON SCREEN. The failure message's most useful sentence. @@ -415,7 +500,25 @@ async def _check(self, cond: Condition, label: str) -> Optional[str]: # state's row names. Kept as a plain visibility check rather than a # new mechanism: the CSD already had to name a tag per state, and a # second way to say the same thing is how two spellings drift. - pass # asserted via `visible:`/`absent:` alongside; see CSD/3 §2.3 + # + # LOCAL DELTA. Upstream this was `pass`, so `state:` asserted + # NOTHING — a vacuous green in the one predicate CSD/3 makes + # mandatory. It now reads the CSD's `states:` map: that state's tag + # must be on screen and every OTHER state's tag must not be, so an + # error cannot pass for an empty list. With no map it FAILS rather + # than passing: an unanchored `state:` is unchecked, not true. + want = self.state_tags.get(cond.state) + if not want: + return (f"{label}: `state: {cond.state}` cannot be checked — no CSD " + f"`states:` tag for it (does the flow name its `csd:`?)") + if not await self.helper.is_element_visible(want): + await self.helper.scroll_into_view(want) + if not await self.helper.is_element_visible(want): + return f"{label}: state {cond.state!r} — its tag {want!r} is not on screen" + for other, tag in sorted(self.state_tags.items()): + if other != cond.state and tag != want and await self.helper.is_element_visible(tag): + return (f"{label}: state {cond.state!r} expected, but {other!r}'s " + f"tag {tag!r} is on screen") if cond.count is not None: got = _match_glob(on_screen, cond.count["of"]) @@ -551,6 +654,10 @@ def _shot(self, spec: FlowSpec, step: Step) -> Optional[str]: return str(got) if got else None async def run(self, spec: FlowSpec) -> bool: + if spec.csd is not None: + # LOCAL DELTA: a flow that names its CSD carries its own maps. + self.field_tags = self.field_tags or dict(spec.csd.field_tags) + self.state_tags = self.state_tags or dict(spec.csd.state_tags) print(f"\n FLOW {spec.flow} — {spec.title}") if spec.description: print(f" {spec.description}") @@ -630,6 +737,7 @@ def write_report(self, spec: FlowSpec) -> Optional[Path]: "flow": spec.flow, "title": spec.title, "spec": str(spec.path), + "csd": spec.csd_id, "client_floor": spec.client_floor, "passed": all(r.status != "fail" for r in self.results), "steps": [ diff --git a/testing/gate/run_flows.py b/testing/gate/run_flows.py new file mode 100644 index 00000000..bc5010b4 --- /dev/null +++ b/testing/gate/run_flows.py @@ -0,0 +1,335 @@ +"""Run this repo's CSD flows against a live client — the `testable` half of CSD/3. + + # against a client that is already running (a desktop at a keyboard): + python3 -m testing.gate.run_flows --platform desktop --flows testing/flows + + # on the matrix: the same code, called by run_platform after its smoke walk, + # in the same process and against the app that walk just brought up: + python3 -m testing.gate.run_platform --platform desktop --jar … --flows testing/flows + +CSD.md §1: a CSD reaches `testable` when "floor flips off unreleased; flow runs on +the matrix". This is the thing that runs it. + +FOUR VERDICTS PER FLOW, AND ONLY TWO OF THEM ARE GREEN. + + pass every step's requires/do/expect held + refused the flow's `client:` floor is above the client under test. The + flow cannot start HERE, and says why; it is neither passed nor + failed, and it does not redden the leg + cannot-start the floor is met, but the flow never reached its first screen: + nav_map has no hop to it on this build, a hop tag was missing + mid-walk, or the first `requires` still did not hold + +THE RUNNER WALKS TO THE FIRST SCREEN. Sign-in lands on Contacts; a flow for any +other surface starts elsewhere. Before step one, the runner clicks the hop +`nav_map` derives for the flow's first `requires: screen:` (circle, tab, row), +waiting for each tag. The flow never encodes the hop (FSD/CSD_STANDARD.md §5). +Flow-only screens (pre-login, wizards, leaves) have no hop and are waited for. + fail it started and a step broke + +`cannot-start` REDDENS THE LEG, deliberately, and differs from CIRISAgent's gate +here. Upstream reports it as a warning because it was their only way to hold a +flow for a surface no release carried. This repo has the `client:` floor for that +(`unreleased`, `>X`). With the floor met, a flow that never reached its first +screen is a flow that silently never ran — and a leg that stays green while its +only flow never ran is the vacuous green this harness exists to refuse. + +LOADING IS ALL-OR-NOTHING, AND IT HAPPENS FIRST. Every flow must parse, name its +CSD, and name no `proposed:` tag before anything is driven: a spec error found +after ten minutes of emulator is ten minutes wasted, and one found after a +partial run hides behind the flows that did run. +""" + +from __future__ import annotations + +import argparse +import asyncio +import re +import json +import sys +import time +from dataclasses import asdict, dataclass, field +from pathlib import Path +from typing import Any, List, Optional, Sequence + +from testing.gate.flow_spec import FlowRunner, FlowSpec, SpecError, check_client_floor, discover + +REPO = Path(__file__).resolve().parents[2] +DEFAULT_FLOWS = REPO / "testing" / "flows" + +PASS, FAIL, REFUSED, CANNOT_START = "pass", "fail", "refused", "cannot-start" +#: The only verdicts that leave a leg green. +GREEN = {PASS, REFUSED} + +#: Screens a signed-out client can be on. Anything else is taken as signed in. +SIGNED_OUT = {"Login", "Setup"} + + +@dataclass +class FlowOutcome: + flow: str + csd: Optional[str] + status: str + detail: str = "" + steps: List[dict] = field(default_factory=list) + report: Optional[str] = None + + +def default_client_version() -> str: + """This tree's version. On the matrix the artifact IS this tree's (asserted + by candidate_artifacts), so the floor is checked against the candidate.""" + try: + return (REPO / "VERSION").read_text(encoding="utf-8").strip() + except OSError: + return "" + + +def load_flows(paths: Sequence[str | Path], csd_root: Optional[Path] = None) -> List[FlowSpec]: + """Every flow, loaded and bound to its CSD — or a SpecError naming the first + that is not. Never a partial list.""" + files = discover([str(p) for p in paths]) + if not files: + raise SpecError(f"no flows found in {[str(p) for p in paths]}") + specs: List[FlowSpec] = [] + seen: dict[str, Path] = {} + for path in files: + spec = FlowSpec.load(path, csd_root=csd_root) + if spec.csd is None: + raise SpecError( + f"{path}: names no `csd:`. Every flow in this repo tests a CSD, and the " + f"runner reads that CSD's `shows:` and `states:` to check it" + ) + if spec.flow in seen: + raise SpecError(f"{path}: flow id {spec.flow!r} is also used by {seen[spec.flow]}") + seen[spec.flow] = path + specs.append(spec) + return specs + + +async def _settle_on(helper, screen: str, timeout: float, poll: float = 1.0) -> str: + """Wait for the flow's starting screen. Not navigation — a landing that is + still composing is not a flow that cannot start. Returns the last screen.""" + deadline = time.monotonic() + timeout + cur = await helper.get_screen() + while cur != screen and time.monotonic() < deadline: + await asyncio.sleep(poll) + cur = await helper.get_screen() + return cur + + +async def navigate(helper, screen: str, chain: Sequence[str], *, hop_timeout: float = 20.0, + arrive_timeout: float = 20.0) -> Optional[str]: + """Walk `chain` (nav_map's derived hop) to `screen`. None on arrival, else + the reason — naming the hop tag that was missing, because "could not reach + Screen.X" alone sends the reader to the wrong end of the chain. + + Each tag is WAITED for before it is clicked: a circle's tabs compose after + the circle is chosen, and clicking before they exist is a race, not a test. + """ + for i, tag in enumerate(chain, 1): + where = f"hop {i} of {len(chain)} ({' -> '.join(chain)})" + if not await helper.wait_for_element(tag, timeout=int(hop_timeout * 1000)): + return (f"navigation to Screen.{screen}: hop tag {tag!r} never appeared, {where}; " + f"on {await helper.get_screen()!r}") + if not await helper.click(tag, timeout=int(hop_timeout * 1000)): + return f"navigation to Screen.{screen}: clicking hop tag {tag!r} failed, {where}" + got = await _settle_on(helper, screen, arrive_timeout) + if got == "CircleTab" and screen != "CircleTab": + # A tab with ONE card opens it directly in the wide layout (nav_map + # drops the row hop), but the compact layout (phones) lists the one + # card first. Open it when there is exactly one row; otherwise name + # the rows rather than guess. + rows = sorted({e.test_tag for e in await helper.get_elements() + if e.test_tag.startswith("nav_epistemic_")}) + # The row for THIS screen, when the list has it (Contacts sits in every + # circle's People tab; a tab can list several cards). + want = "nav_epistemic_" + re.sub(r"(? '.join(chain)} and landed on " + f"the tab's card list with {len(rows)} rows ({', '.join(rows)}); nav_map " + f"expected one card, and none is {want!r} — was the circle hop applied?") + if got != screen: + return (f"navigation to Screen.{screen}: walked {' -> '.join(chain)} and landed on " + f"{got!r}") + return None + + +def nav_hops(has_agent: bool) -> tuple[dict, set]: + """(Screen -> hop, flow-only screens) for this build, from the client source.""" + from testing.gate import nav_map, screen_atlas # noqa: PLC0415 + return nav_map.build(has_agent=has_agent), screen_atlas.flow_only() + + +async def run_one(spec: FlowSpec, helper, *, platform=None, artifacts: Optional[Path] = None, + client_version: Optional[str] = None, start_timeout: float = 30.0, + hops: Optional[dict] = None, flow_only: frozenset | set = frozenset()) -> FlowOutcome: + """Run one flow. With `hops` (nav_map's Screen -> chain), the runner first + WALKS to the flow's starting screen; without, it only waits for it.""" + refusal = check_client_floor(spec.client_floor, client_version) + if refusal: + print(f"\n FLOW {spec.flow} ({spec.csd_id}) — REFUSED by its floor\n {refusal}") + return FlowOutcome(spec.flow, spec.csd_id, REFUSED, refusal) + + start = spec.steps[0].requires.screen + if start and hops is None: + await _settle_on(helper, start, start_timeout) + elif start: + # A landing still composing is not a flow on the wrong screen: give the + # client a moment before deciding to walk anywhere. + cur = await _settle_on(helper, start, min(start_timeout, 5.0)) + err = None + if cur != start: + if start in hops: + print(f"\n FLOW {spec.flow} — walking to Screen.{start}: {' -> '.join(hops[start])}") + err = await navigate(helper, start, hops[start], + hop_timeout=start_timeout, arrive_timeout=start_timeout) + elif start in flow_only: + # Pre-login, wizards, leaves: nothing in the shell leads there, + # so the flow must already be on it. Wait, then let `requires` judge. + await _settle_on(helper, start, start_timeout) + else: + err = (f"no nav hop for Screen.{start} on this build, and it is not a " + f"flow-only screen (on {cur!r})") + if err: + print(f"\n FLOW {spec.flow} ({spec.csd_id}) — CANNOT START\n {err}") + return FlowOutcome(spec.flow, spec.csd_id, CANNOT_START, err) + + runner = FlowRunner(helper, platform=platform, artifacts=artifacts) + try: + ok = await runner.run(spec) + except Exception as e: # noqa: BLE001 — a crash in a flow is that flow's verdict + ok = False + runner.results.append(_crash(e)) + report = runner.write_report(spec) + steps = [asdict(r) for r in runner.results] + print(f"\n {runner.summary(spec)}") + + if ok: + status, detail = PASS, runner.summary(spec) + else: + bad = next((r for r in reversed(runner.results) if r.status == "fail"), None) + first = runner.results[0] if runner.results else None + if first is not None and len(runner.results) == 1 and first.phase == "requires": + status, detail = CANNOT_START, f"first step {first.step_id!r}: {first.detail}" + else: + status = FAIL + detail = f"step {bad.step_id!r} ({bad.phase}): {bad.detail}" if bad else "failed" + return FlowOutcome(spec.flow, spec.csd_id, status, detail, steps, str(report) if report else None) + + +def _crash(e: Exception): + from testing.gate.flow_spec import StepResult # noqa: PLC0415 + return StepResult("", "the runner raised", "fail", "do", f"{type(e).__name__}: {e}") + + +def leg_ok(outcomes: Sequence[FlowOutcome]) -> bool: + return all(o.status in GREEN for o in outcomes) + + +def summary(outcomes: Sequence[FlowOutcome]) -> str: + counts = {s: sum(1 for o in outcomes if o.status == s) for s in (PASS, FAIL, CANNOT_START, REFUSED)} + return ", ".join(f"{n} {s}" for s, n in counts.items() if n) or "no flows" + + +def sign_in(drv, username: str, password: str) -> str: + """Reuse the session if the client has one; make one if it does not.""" + from testing.gate import session_fixture # noqa: PLC0415 + + screen = drv.screen() + if screen not in SIGNED_OUT: + return screen + session_fixture.run_setup(drv, username, password) + return session_fixture.log_in(drv, username, password) + + +def run_all(specs: Sequence[FlowSpec], drv, *, platform=None, artifacts: Optional[Path] = None, + client_version: Optional[str] = None, username: str = "qaadmin", + password: str = "QaAdmin!2345", establish_session: bool = True, + helper: Any = None, navigate_to_start: bool = True) -> List[FlowOutcome]: + """Run every flow in order against one live client. Signs in once, only if + some flow will actually run.""" + from testing.gate.flow_helper import SyncFlowHelper # noqa: PLC0415 + + helper = helper or SyncFlowHelper(drv) + runnable = [s for s in specs if not check_client_floor(s.client_floor, client_version)] + if runnable and establish_session and drv is not None: + landed = sign_in(drv, username, password) + print(f" session: signed in, on {landed!r}") + + hops, flow_only = None, frozenset() + if runnable and navigate_to_start: + # The build decides the tree: a node client has no agentOnly rows, so a + # hop derived for the agent build would click tags that are not there. + mode = "" + if drv is not None: + try: + mode = str(drv.state().get("clientMode", "")) + except Exception: # noqa: BLE001 — unknown mode: the node tree, the subset + mode = "" + hops, flow_only = nav_hops(has_agent=mode.upper() == "AGENT") + + async def go() -> List[FlowOutcome]: + return [await run_one(s, helper, platform=platform, artifacts=artifacts, + client_version=client_version, hops=hops, + flow_only=flow_only) for s in specs] + + outcomes = asyncio.run(go()) + print(f"\n flows: {summary(outcomes)}") + for o in outcomes: + print(f" [{o.status:^12}] {o.flow} ({o.csd}): {o.detail}") + return outcomes + + +def main(argv: Optional[List[str]] = None) -> int: + ap = argparse.ArgumentParser(description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("--platform", default="desktop", choices=("desktop", "android", "ios")) + ap.add_argument("--url", default="http://127.0.0.1:9091", help="the client's test server") + ap.add_argument("--flows", action="append", help=f"a flow or a directory (default {DEFAULT_FLOWS})") + ap.add_argument("--csd-root", type=Path, help="where the CSDs live (default FSD/CSD)") + ap.add_argument("--client-version", default=None, + help="the version under test, for `client:` floors (default: VERSION)") + ap.add_argument("--artifacts", type=Path, help="screenshots and per-flow JSON") + ap.add_argument("--report", type=Path, help="write every outcome here as JSON") + ap.add_argument("--username", default="qaadmin") + ap.add_argument("--password", default="QaAdmin!2345") + ap.add_argument("--no-sign-in", action="store_true", + help="drive whatever screen the client is on; do not make a session") + args = ap.parse_args(argv) + + try: + specs = load_flows(args.flows or [DEFAULT_FLOWS], args.csd_root) + except SpecError as e: + print(f"[FAIL] {e}") + return 1 + print(f"loaded {len(specs)} flow(s): {', '.join(f'{s.flow} ({s.csd_id})' for s in specs)}") + + from testing.driver import DriverError, TestAutomationServer # noqa: PLC0415 + from testing.gate.platforms import build_platform # noqa: PLC0415 + from testing.gate.session_fixture import SessionUnavailable # noqa: PLC0415 + + drv = TestAutomationServer(base_url=args.url) + version = args.client_version if args.client_version is not None else default_client_version() + artifacts = args.artifacts or Path("shots") / f"flows-{args.platform}" + try: + drv.wait_for_server(timeout=30) + outcomes = run_all(specs, drv, platform=build_platform(args), artifacts=artifacts, + client_version=version, username=args.username, + password=args.password, establish_session=not args.no_sign_in) + except (DriverError, SessionUnavailable) as e: + print(f"[FAIL] {e}") + return 1 + if args.report: + args.report.parent.mkdir(parents=True, exist_ok=True) + args.report.write_text(json.dumps([asdict(o) for o in outcomes], indent=2), encoding="utf-8") + ok = leg_ok(outcomes) + print(f"flows: {'PASS' if ok else 'FAIL'}") + return 0 if ok else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/testing/gate/run_platform.py b/testing/gate/run_platform.py index 1f0dc384..b78db601 100644 --- a/testing/gate/run_platform.py +++ b/testing/gate/run_platform.py @@ -62,6 +62,8 @@ class Report: artifact: str = "" node_version: str = "" steps: list[StepResult] = field(default_factory=list) + #: One entry per CSD flow (run_flows.FlowOutcome), when `--flows` was given. + flows: list[dict] = field(default_factory=list) def add(self, name: str, ok: bool, detail: str = "", shot: str | None = None) -> None: self.steps.append(StepResult(name, ok, detail, shot)) @@ -247,6 +249,11 @@ def main() -> int: ap.add_argument("--report", type=Path) ap.add_argument("--node-version", default="") ap.add_argument("--timeout", type=float, default=120.0) + ap.add_argument("--flows", action="append", + help="after the smoke walk, run these CSD flows (a file or a directory; " + "repeatable) against the same app — see testing/flows/README.md") + ap.add_argument("--client-version", default=None, + help="the version under test, for flows' `client:` floors (default: VERSION)") args = ap.parse_args() rep = Report(platform=args.platform, node_version=args.node_version) @@ -255,6 +262,18 @@ def main() -> int: from testing.gate.platforms import build_platform platform = build_platform(args) + # FLOWS LOAD BEFORE ANYTHING BOOTS. A flow that does not parse, or names a + # CSD that does not, is found in a second — not after the emulator. + specs = None + if args.flows: + from testing.gate import run_flows + from testing.gate.flow_spec import SpecError + try: + specs = run_flows.load_flows(args.flows) + except SpecError as e: + rep.add("flows-load", False, str(e)) + return _finish(args, rep) + plan = None try: plan = plan_for(args) @@ -266,6 +285,8 @@ def main() -> int: # PROVEN, NOT ASSUMED. drv.wait_for_server(timeout=args.timeout) walk(drv, rep, args.shots, platform, args_timeout=args.timeout) + if specs is not None: + flows(drv, rep, specs, args, platform) rep.ok = all(s.ok for s in rep.steps) except bringup.CannotRun as e: # LOUD. Not a skip: the caller decides what to exclude, and it does so @@ -288,7 +309,39 @@ def main() -> int: # check=False: teardown runs after failures too, and one that fails # must not hide the failure that caused it. bringup.run(td, check=False) + return _finish(args, rep) + +def flows(drv: TestAutomationServer, rep: Report, specs, args, platform) -> None: + """The CSD flows, in the SAME app the walk just proved — no second bring-up. + + Only after a clean walk: the walk is what proves there is a composed app on + a real node to drive, and flows run on anything less would fail for the + walk's reason under their own names. + """ + from testing.gate import run_flows + from testing.gate.session_fixture import SessionUnavailable + + if not all(s.ok for s in rep.steps): + rep.add("flows", False, "not run — the smoke walk above failed, so there is no app to drive") + return + # Per LEG, not per platform: the linux desktop and the android emulator + # share a runner and a `shots/` directory, and must not overwrite each other. + leg = args.report.stem if args.report else args.platform + version = args.client_version if args.client_version is not None else run_flows.default_client_version() + try: + outcomes = run_flows.run_all(specs, drv, platform=platform, + artifacts=args.shots / f"flows-{leg}", + client_version=version) + except SessionUnavailable as e: + rep.add("flows", False, f"no session to run them in: {e}") + return + rep.flows = [asdict(o) for o in outcomes] + rep.add("flows", run_flows.leg_ok(outcomes), run_flows.summary(outcomes)) + + +def _finish(args, rep: Report) -> int: + rep.ok = rep.ok and all(s.ok for s in rep.steps) if args.report: # MAKE THE DIRECTORY. `--report reports/.json` names a path in a # directory nothing creates: the workflow passes it, the artifact upload diff --git a/testing/gate/session_fixture.py b/testing/gate/session_fixture.py index 62e2aa20..078428b1 100644 --- a/testing/gate/session_fixture.py +++ b/testing/gate/session_fixture.py @@ -64,6 +64,71 @@ def _tags(drv: TestAutomationServer) -> set[str]: return {e.test_tag for e in drv.tree()} +def _wait_clickable(drv: TestAutomationServer, tag: str, timeout: float = 20.0) -> bool: + """True once `tag` reports can_click (or the server does not report it at all).""" + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + for e in drv.tree(): + if e.test_tag == tag: + if e.can_click is None or e.can_click: + return True + break + time.sleep(1.0) + return False + + +def _reach(drv: TestAutomationServer, tag: str, act, tries: int = 8): + """Run `act()` against `tag`, scrolling it into view first when the app + says it is composed but off screen. A phone's first-run wizard is taller + than the screen: on the iPhone the account fields sit below the age band + and the AI choice, and /input and /click refuse what the person could not + see (CIRISClient#33). Scroll down a step at a time, bounded; anything else + raises as before.""" + # Down first (the wizard fills top to bottom), then back up past the + # start: an element the earlier steps scrolled past sits ABOVE the fold. + notes: list[str] = [] + # Down until the screen says it is at the bottom, then up until the top, + # trying the act after every step. A phone with the keyboard up has a + # small viewport, so a fixed number of steps can turn around before it + # ever reaches the last field. + for direction in ("down", "up"): + for _ in range(tries * 3): + try: + return act() + except DriverError as e: + if "off screen" not in str(e): + raise + try: + r = drv.scroll_to(tag, direction=direction, amount=300) + msg = (r or {}).get("error") if isinstance(r, dict) else None + except DriverError as se: + msg = str(se)[-160:] + notes.append(f"{direction}: {msg or 'moved'}") + time.sleep(0.6) + if msg and ("already at the" in msg or "NO overflow" in msg): + break + try: + return act() + except DriverError as e: + # Say what the scrolls answered: "no overflow" means the wizard's own + # scrollable is not the one registered, which is a client defect. + raise DriverError(f"{e} | scrolls: {'; '.join(dict.fromkeys(notes))}") from None + + +def _field_report(drv: TestAutomationServer) -> str: + """What each tagged element on screen holds: input values (passwords by + length only), texts, and click/input capability — so a disabled Next + names the condition it is waiting on instead of just the tag list.""" + parts = [] + for e in drv.tree(): + val = getattr(e, "input_value", None) + if val is not None and "password" in e.test_tag: + val = f"<{len(val)} chars>" + txt = (e.text or "")[:60] + parts.append(f"{e.test_tag}[v={val!r} t={txt!r} c={e.can_click} i={e.can_input}]") + return "; ".join(sorted(parts)) + + def _settle(drv: TestAutomationServer, want: str, timeout: float = 90.0) -> bool: deadline = time.monotonic() + timeout while time.monotonic() < deadline: @@ -79,7 +144,11 @@ def run_setup(drv: TestAutomationServer, username: str, password: str, if "txt_owner_hint" in _tags(drv): return # already owned; nothing to do - drv.click("btn_local_login") + # Desktop's first run shows Login; a client whose first run opens the wizard + # directly is already where this click would take it, and clicking a + # `btn_local_login` that is not on screen fails the fixture for nothing. + if drv.screen() != "Setup": + drv.click("btn_local_login") if not _settle(drv, "Setup", timeout=30): raise SessionUnavailable( f"btn_local_login did not reach Setup (on {drv.screen()!r}); on a node " @@ -91,10 +160,33 @@ def run_setup(drv: TestAutomationServer, username: str, password: str, ("input_password_confirm", password), ("input_device_name", device)): try: - drv.input(tag, value) + _reach(drv, tag, lambda t=tag, v=value: drv.input(t, v)) except DriverError as e: raise SessionUnavailable(f"wizard: {tag} would not accept input ({e})") from e - drv.click("age_band_adult") + # iOS needs ~2 s between fields for the value to reach the ViewModel's + # StateFlow (client/CLAUDE.md, "Important iOS notes"). Without it the + # fields read empty and Next never enables: the first iOS run of this + # fixture stopped at "wizard did not advance past 'you'". + time.sleep(2.0) + # The fed-ID label: desktop mints/admits the identity itself, but a client + # that asks (iOS) keeps Next disabled until the label is valid + # (SetupState.canProceedFromCurrentStep: YOU -> fedIdOk; generic words + # like "me" are refused, so use the device name, which is specific). + if "input_fedid_label" in _tags(drv): + try: + _reach(drv, "input_fedid_label", + lambda: drv.input("input_fedid_label", f"{device} gate identity")) + except DriverError as e: + raise SessionUnavailable(f"wizard: input_fedid_label would not accept input ({e})") from e + time.sleep(2.0) + _reach(drv, "age_band_adult", lambda: drv.click("age_band_adult")) + # The legs run against a bare node with no LLM, so answer "run without AI": + # it removes the AI step, whose Next waits for a usable LLM choice + # (SetupState: AI -> hasUsableLlmChoice). Desktop already defaults there; + # iOS asks, and the walk stopped at 'ai' with Next disabled. + if "opt_run_without_ai" in _tags(drv): + _reach(drv, "opt_run_without_ai", lambda: drv.click("opt_run_without_ai")) + time.sleep(1.0) # Advance until the claim takes over. Bounded: a wizard that stops advancing # must say so rather than spin. @@ -106,7 +198,12 @@ def run_setup(drv: TestAutomationServer, username: str, password: str, # answered (no default, like the age band above). Yes is the fixture's # answer for the same reason age_band_adult is: the unrestricted path. if "trace_consent_yes" in tags: - drv.click("trace_consent_yes") + _reach(drv, "trace_consent_yes", lambda: drv.click("trace_consent_yes")) + # The step advances only once the answer has reached the ViewModel + # (`SetupState.canProceedFromCurrentStep`: JOIN_FEDERATION -> + # traceConsentAnswered). A Next clicked in the same instant as the + # answer raced it on Windows; give the answer a beat to land. + time.sleep(1.5) tags = _tags(drv) nxt = "btn_wizard_complete" if "btn_wizard_complete" in tags else "btn_next" if nxt not in tags: @@ -114,11 +211,51 @@ def run_setup(drv: TestAutomationServer, username: str, password: str, f"wizard step {_active_step(drv)!r} offers no advance control; " f"on screen: {sorted(tags)}" ) + # A disabled advance control is present but has no click handler (a + # `testableClickable(enabled = false)`), and clicking it is a 404 that + # says nothing about WHY. On iOS the fields reach the ViewModel a beat + # after they are typed, so Next enables late: wait for it, bounded, and + # if it never enables say which step and what was on screen. + if not _wait_clickable(drv, nxt, timeout=20.0): + raise SessionUnavailable( + f"wizard step {_active_step(drv)!r}: {nxt} never became clickable; " + f"on screen: {sorted(_tags(drv))}" + ) before = (drv.screen(), _active_step(drv)) - drv.click(nxt) + # iOS's /tree omits `canClick` when it is false (defaults are not + # serialized), so `_wait_clickable` cannot see a disabled Next there. + # A disabled control answers the click with 404 "No click handler": + # treat that as "not yet", bounded, and name the step if it stays so. + clicked = False + for _ in range(15): + try: + _reach(drv, nxt, lambda: drv.click(nxt)) + clicked = True + break + except DriverError as e: + if "No click handler" not in str(e): + raise + time.sleep(2.0) + if not clicked: + raise SessionUnavailable( + f"wizard step {before[1]!r}: {nxt} stayed disabled for 30s; " + f"fields: {_field_report(drv)}" + ) time.sleep(2.0) if (drv.screen(), _active_step(drv)) == before and "setup_ownership_claimed" not in _tags(drv): - raise SessionUnavailable(f"wizard did not advance past {before[1]!r}") + # One retry when the step's question is still on screen: the answer + # may not have landed before Next was clicked. + if "trace_consent_yes" in _tags(drv): + drv.click("trace_consent_yes") + time.sleep(2.0) + drv.click(nxt) + time.sleep(2.0) + if (drv.screen(), _active_step(drv)) == before and "setup_ownership_claimed" not in _tags(drv): + # Say what was on screen: a required field the fixture doesn't fill + # (the with-AI wizard asks for more than the node one) shows up here. + raise SessionUnavailable( + f"wizard did not advance past {before[1]!r}; on screen: {sorted(_tags(drv))}" + ) # The claim has no button; it completes and the app returns to Login. if not _settle(drv, "Login", timeout=180): diff --git a/testing/test_driver_rules.py b/testing/test_driver_rules.py index c483c6a6..ad890566 100644 --- a/testing/test_driver_rules.py +++ b/testing/test_driver_rules.py @@ -206,3 +206,31 @@ def test_wait_for_ui_does_not_confuse_a_live_server_for_a_composed_app(server): assert drv.wait_for_ui(timeout=1.0) == 0, ( "a reachable automation server was mistaken for a composed app" ) + + +def test_a_body_that_says_it_failed_raises_even_on_http_200(): + """iOS answered a failed /input with 200 {"success": false}; the walk then + 'typed' into fields that took nothing (CSD flows on the matrix, #97).""" + import http.server, threading, json as _json + from testing.driver import TestAutomationServer, DriverError + + class H(http.server.BaseHTTPRequestHandler): + def do_POST(self): + n = int(self.headers.get("Content-Length", 0)); self.rfile.read(n) + body = _json.dumps({"success": False, "error": "no text sink is listening for input_x"}).encode() + self.send_response(200); self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(body))); self.end_headers(); self.wfile.write(body) + def log_message(self, *a): pass + + srv = http.server.HTTPServer(("127.0.0.1", 0), H) + t = threading.Thread(target=srv.serve_forever, daemon=True); t.start() + try: + drv = TestAutomationServer(base_url=f"http://127.0.0.1:{srv.server_address[1]}") + try: + drv.input("input_x", "hello", verify=False) + except DriverError as e: + assert "no text sink" in str(e) + else: + raise AssertionError("a success:false body was accepted as typed") + finally: + srv.shutdown() diff --git a/testing/test_flows.py b/testing/test_flows.py new file mode 100644 index 00000000..ee31e20d --- /dev/null +++ b/testing/test_flows.py @@ -0,0 +1,595 @@ +"""CSD flows on the matrix: loading, the floor, and the runner's verdicts. + +Every rule here has a red path, and each test is the red path — a check whose +failing half has never run is a check with an untested half (AGENTS.md). The +runner tests drive `run_flows.run_one` through a FAKE helper, which is honest +here in a way it is not for the driver: what is under test is the verdict logic +over `/tree`-shaped answers, not the transport (test_driver_rules.py owns that). +""" + +from __future__ import annotations + +import asyncio +import re +import textwrap +from dataclasses import dataclass +from pathlib import Path +from typing import Dict, Optional + +import pytest +import yaml + +from testing.gate import run_flows +from testing.gate.flow_spec import FlowSpec, SpecError + +REPO = Path(__file__).resolve().parents[1] +FLOWS = REPO / "testing" / "flows" + +CSD_TEXT = """\ +# CSD-900 — a test surface + +```yaml csd:stage +stage: sketched +owner: CIRISClient +``` + +```yaml csd:shows +fields: + - ceg: x_private:score + use: display-only + type: float + example: 0.5 + renders: "0.5" + tag: value_score + - ceg: x_private:chip + use: display-only + type: string + example: "a" + renders: "a chip" + tag: "proposed:row_chip" +``` + +```yaml csd:states +populated: {tag: thing_list} +empty: {tag: thing_empty} +loading: {renders: "a spinner, no tag"} +error: {tag: "proposed:thing_error"} +``` +""" + + +def _csd_root(tmp_path: Path, text: str = CSD_TEXT, name: str = "CSD-900-test.md") -> Path: + root = tmp_path / "csd" + root.mkdir(exist_ok=True) + (root / name).write_text(text, encoding="utf-8") + return root + + +def _flow(tmp_path: Path, body: str, name: str = "f.yaml") -> Path: + p = tmp_path / name + p.write_text(textwrap.dedent(body), encoding="utf-8") + return p + + +GOOD = """\ + flow: thing + csd: CSD-900 + client: ">=0.5.224" + steps: + - step_id: land + title: lands + requires: {screen: Thing} + expect: + visible: [thing_list] +""" + + +# ── loading ───────────────────────────────────────────────────────────────── + +def test_a_good_flow_loads_and_carries_its_csds_maps(tmp_path): + spec = FlowSpec.load(_flow(tmp_path, GOOD), csd_root=_csd_root(tmp_path)) + assert spec.csd_id == "CSD-900" + assert spec.csd.field_tags["x_private:score"] == "value_score" + assert spec.csd.state_tags == {"populated": "thing_list", "empty": "thing_empty"} + assert "row_chip" in spec.csd.proposed and "thing_error" in spec.csd.proposed + + +def test_an_unknown_top_level_key_is_a_load_error(tmp_path): + p = _flow(tmp_path, GOOD + " csd_id: CSD-900\n") + with pytest.raises(SpecError, match="unknown key"): + FlowSpec.load(p, csd_root=_csd_root(tmp_path)) + + +def test_a_csd_that_does_not_exist_is_a_load_error_not_a_skip(tmp_path): + p = _flow(tmp_path, GOOD.replace("CSD-900", "CSD-999")) + with pytest.raises(SpecError, match="does not exist"): + FlowSpec.load(p, csd_root=_csd_root(tmp_path)) + + +def test_a_malformed_csd_id_is_a_load_error(tmp_path): + p = _flow(tmp_path, GOOD.replace("CSD-900", "people")) + with pytest.raises(SpecError, match="not a CSD id"): + FlowSpec.load(p, csd_root=_csd_root(tmp_path)) + + +def test_a_csd_whose_blocks_do_not_parse_is_a_load_error(tmp_path): + broken = CSD_TEXT.replace("populated: {tag: thing_list}", "populated: {tag: [unclosed") + with pytest.raises(SpecError, match="does not parse"): + FlowSpec.load(_flow(tmp_path, GOOD), csd_root=_csd_root(tmp_path, broken)) + + +def test_a_document_with_no_stage_block_is_not_a_csd(tmp_path): + prose = "# CSD-900\n\nJust prose, no typed blocks.\n" + with pytest.raises(SpecError, match="not a CSD/3 document"): + FlowSpec.load(_flow(tmp_path, GOOD), csd_root=_csd_root(tmp_path, prose)) + + +def test_a_flow_in_this_repo_must_name_its_csd(tmp_path): + no_csd = GOOD.replace(" csd: CSD-900\n", "") + d = tmp_path / "flows" + d.mkdir() + _flow(d, no_csd) + with pytest.raises(SpecError, match="names no `csd:`"): + run_flows.load_flows([d], csd_root=_csd_root(tmp_path)) + + +def test_an_empty_flows_directory_is_a_load_error(tmp_path): + d = tmp_path / "flows" + d.mkdir() + with pytest.raises(SpecError, match="no flows found"): + run_flows.load_flows([d]) + + +@pytest.mark.parametrize("where", [ + " expect:\n visible: [row_chip]\n", + " expect:\n absent: [row_chip]\n", + " expect:\n text: {row_chip: a}\n", + " do:\n - click: row_chip\n", + " requires:\n visible: [row_chip]\n expect:\n visible: [thing_list]\n", +]) +def test_a_proposed_tag_named_anywhere_in_a_flow_is_a_load_error(tmp_path, where): + body = GOOD.split(" - step_id")[0] + " - step_id: s\n title: t\n" + where + with pytest.raises(SpecError, match="proposed"): + FlowSpec.load(_flow(tmp_path, body), csd_root=_csd_root(tmp_path)) + + +def _a_real_proposed_tag(): + """(CSD id, a tag it still marks `proposed:`) from the shipped documents. + + Derived, not named: this test used CSD-005's `contacts_row_trust`, and the + chip shipped — the document stopped marking it proposed and the test went + red for the product doing its job. Any real CSD with a proposed tag proves + the same thing.""" + import re as _re # noqa: PLC0415 + from testing.gate.csd_doc import CsdError, load # noqa: PLC0415 + for path in sorted((REPO / "FSD" / "CSD").glob("CSD-*.md")): + m = _re.match(r"(CSD-\d+)", path.name) + try: + doc = load(m.group(1)) if m else None + except CsdError: + continue + if doc is not None and doc.proposed: + return doc.csd_id, sorted(doc.proposed)[0] + return None + + +def test_a_real_csd_refuses_a_tag_it_still_marks_proposed(tmp_path): + """Against a shipped document, not a fixture: a tag a real CSD still marks + `proposed:` may not be asserted by a flow.""" + found = _a_real_proposed_tag() + if found is None: + pytest.skip("no shipped CSD marks any tag proposed") + csd_id, tag = found + body = f"""\ + flow: p + csd: {csd_id} + steps: + - step_id: s + title: t + expect: + visible: [{tag}] + """ + with pytest.raises(SpecError, match=f"{tag}.*proposed"): + FlowSpec.load(_flow(tmp_path, body)) + + +def test_a_state_whose_tag_is_proposed_is_a_load_error(tmp_path): + body = GOOD.replace("visible: [thing_list]", "state: error") + with pytest.raises(SpecError, match="state: error.*proposed"): + FlowSpec.load(_flow(tmp_path, body), csd_root=_csd_root(tmp_path)) + + +def test_a_state_the_csd_gives_no_tag_is_a_load_error(tmp_path): + body = GOOD.replace("visible: [thing_list]", "state: loading") + with pytest.raises(SpecError, match="names no tag"): + FlowSpec.load(_flow(tmp_path, body), csd_root=_csd_root(tmp_path)) + + +def test_a_relation_over_a_field_the_csd_does_not_show_is_a_load_error(tmp_path): + body = GOOD.replace( + "visible: [thing_list]", + "relation: {left: 'x_private:score', op: eq, right: 'x_private:nope'}") + with pytest.raises(SpecError, match="not a field"): + FlowSpec.load(_flow(tmp_path, body), csd_root=_csd_root(tmp_path)) + + +def test_a_flow_without_csd_still_loads_through_flow_spec_alone(tmp_path): + """The delta is additive: an upstream-shaped flow (no `csd:`) loads as before.""" + spec = FlowSpec.load(_flow(tmp_path, GOOD.replace(" csd: CSD-900\n", ""))) + assert spec.csd is None + + +# ── the seeded flows ──────────────────────────────────────────────────────── + +def test_every_flow_in_the_repo_loads_against_its_real_csd(): + specs = run_flows.load_flows([FLOWS]) + assert specs, "testing/flows is empty — the matrix would run nothing and pass" + assert all(s.csd is not None for s in specs) + + +def _client_tag_strings() -> tuple[set[str], set[str]]: + """(whole tag literals, prefixes of interpolated tags) in commonMain. + + `"age_band_$token"` builds `age_band_adult`, so a literal-only read calls a + real tag missing. A prefix counts only if it has two segments + (`age_band_`, not `btn_`): `"btn_$x"` would otherwise vouch for every + button a flow could ever name, which is no check at all. + """ + src = REPO / "client" / "shared" / "src" / "commonMain" + literals: set[str] = set() + prefixes: set[str] = set() + for kt in src.rglob("*.kt"): + for body, end in re.findall(r'"([a-z][a-z0-9_]*)(["$])', kt.read_text(encoding="utf-8")): + if end == '"': + literals.add(body) + elif "_" in body.rstrip("_"): + prefixes.add(body) + return literals, prefixes + + +def client_carries(tag: str, literals: set[str], prefixes: set[str]) -> bool: + return tag in literals or any(tag.startswith(p) for p in prefixes) + + +@pytest.mark.parametrize("tag,carried", [ + ("age_band_adult", True), # "age_band_$token" SetupScreen.kt + ("trace_consent_yes", True), # "trace_consent_$token" SetupScreen.kt + ("radio_cohort_family", True), # "radio_cohort_$value" ClaimNodeScreen.kt + ("chk_duty_box_accept", True), # "chk_duty_box_$verb" DutyConferralScreen.kt + ("opt_run_with_ai", True), # a whole literal still matches + ("contacts_no_such_tag", False), + ("btn_no_such_button", False), # a one-segment prefix vouches for nothing +]) +def test_the_client_tag_check_sees_interpolated_tags_and_nothing_else(tag, carried): + assert client_carries(tag, *_client_tag_strings()) is carried + + +def test_every_tag_a_seeded_flow_names_exists_in_the_client(): + """A flow naming a tag the client does not carry fails as 'element not + found' on every leg. Checked here, at the keyboard, rather than there.""" + literals, prefixes = _client_tag_strings() + missing = [] + for spec in run_flows.load_flows([FLOWS]): + for step in spec.steps: + tags = [a.target for a in step.do] + for cond in (step.requires, step.expect): + tags += cond.visible + cond.absent + list(cond.text) + if cond.state: + tags.append(spec.csd.state_tags[cond.state]) + missing += [f"{spec.flow}/{step.step_id}: {t}" for t in tags + if not client_carries(t, literals, prefixes)] + assert not missing, f"tags no client source carries: {missing}" + + +# ── the runner, over a fake helper ────────────────────────────────────────── + +@dataclass +class _El: + test_tag: str + text: Optional[str] = "" + visible: Optional[bool] = True + width: int = 10 + height: int = 10 + + +class FakeHelper: + """`/tree` as a dict. Records every call so a refusal can be shown to have + driven NOTHING.""" + + def __init__(self, screen: str, tags: Dict[str, str]): + self.screen = screen + self.els = {t: _El(t, txt) for t, txt in tags.items()} + self.calls: list[str] = [] + self.leads: dict = {} + + async def get_elements(self): + self.calls.append("tree") + return list(self.els.values()) + + async def get_element(self, tag): + self.calls.append(f"get {tag}") + return self.els.get(tag) + + async def get_screen(self): + self.calls.append("screen") + return self.screen + + async def is_element_visible(self, tag): + return tag in self.els + + async def scroll_into_view(self, tag): + return tag in self.els + + async def click(self, tag, timeout=2000): + self.calls.append(f"click {tag}") + if tag not in self.els: + return False + # A click can move the app: `leads` maps a tag to (screen, tags now shown). + if tag in self.leads: + self.screen, shown = self.leads[tag] + self.els = {t: _El(t, "") for t in shown} + return True + + async def input_text(self, tag, text): + self.calls.append(f"input {tag}") + return tag in self.els + + async def wait_for_element(self, tag, timeout=2000): + return tag in self.els + + +def _run(spec, helper, version="0.5.224"): + return asyncio.run(run_flows.run_one(spec, helper, client_version=version, start_timeout=0)) + + +def _spec(tmp_path, body=GOOD): + return FlowSpec.load(_flow(tmp_path, body), csd_root=_csd_root(tmp_path)) + + +def test_a_flow_whose_expects_hold_passes(tmp_path): + out = _run(_spec(tmp_path), FakeHelper("Thing", {"thing_list": ""})) + assert out.status == run_flows.PASS + assert run_flows.leg_ok([out]) + + +def test_a_failing_expect_fails_the_flow_and_the_leg(tmp_path): + out = _run(_spec(tmp_path), FakeHelper("Thing", {"something_else": ""})) + assert out.status == run_flows.FAIL + assert "thing_list" in out.detail and "expect" in out.detail + assert out.steps and out.steps[-1]["status"] == "fail" + assert not run_flows.leg_ok([out]) + + +def test_a_flow_that_never_reaches_its_first_screen_cannot_start_and_reddens_the_leg(tmp_path): + out = _run(_spec(tmp_path), FakeHelper("Login", {"thing_list": ""})) + assert out.status == run_flows.CANNOT_START + assert "Thing" in out.detail + assert not run_flows.leg_ok([out]), "a flow that never ran must not leave the leg green" + + +@pytest.mark.parametrize("floor,version", [ + (">=9.9.9", "0.5.224"), + (">0.5.224", "0.5.224"), + ("unreleased", "0.5.224"), +]) +def test_a_flow_above_its_floor_is_refused_neither_passed_nor_failed(tmp_path, floor, version): + spec = _spec(tmp_path, GOOD.replace('">=0.5.224"', f'"{floor}"')) + helper = FakeHelper("Thing", {"thing_list": ""}) + out = _run(spec, helper, version) + assert out.status == run_flows.REFUSED + assert helper.calls == [], "a refused flow must drive nothing" + assert run_flows.leg_ok([out]), "refused is not a failure" + assert "refused" in run_flows.summary([out]) + + +def test_a_met_floor_is_not_refused(tmp_path): + out = _run(_spec(tmp_path), FakeHelper("Thing", {"thing_list": ""}), "0.5.225+preview.gabc") + assert out.status == run_flows.PASS + + +def test_one_red_flow_among_green_ones_reddens_the_leg(tmp_path): + ok = run_flows.FlowOutcome("a", "CSD-900", run_flows.PASS) + refused = run_flows.FlowOutcome("b", "CSD-900", run_flows.REFUSED) + bad = run_flows.FlowOutcome("c", "CSD-900", run_flows.FAIL) + assert run_flows.leg_ok([ok, refused]) + assert not run_flows.leg_ok([ok, refused, bad]) + + +# ── `state:` is checked now, against the CSD's `states:` ──────────────────── + +STATE_FLOW = GOOD.replace("visible: [thing_list]", "state: empty") + + +def test_state_holds_when_its_tag_is_on_screen_and_no_other_states_is(tmp_path): + out = _run(_spec(tmp_path, STATE_FLOW), FakeHelper("Thing", {"thing_empty": ""})) + assert out.status == run_flows.PASS + + +def test_state_fails_when_its_tag_is_not_on_screen(tmp_path): + out = _run(_spec(tmp_path, STATE_FLOW), FakeHelper("Thing", {"other": ""})) + assert out.status == run_flows.FAIL and "thing_empty" in out.detail + + +def test_state_fails_when_another_states_tag_is_also_on_screen(tmp_path): + """Empty and populated at once is not 'empty' — the point of the map.""" + out = _run(_spec(tmp_path, STATE_FLOW), + FakeHelper("Thing", {"thing_empty": "", "thing_list": ""})) + assert out.status == run_flows.FAIL and "populated" in out.detail + + +def test_state_with_no_csd_map_fails_rather_than_passing(tmp_path): + """Upstream's `state:` was a no-op. Unanchored, it now refuses to be green.""" + from testing.gate.flow_spec import FlowRunner + spec = FlowSpec.load(_flow(tmp_path, STATE_FLOW.replace(" csd: CSD-900\n", ""))) + runner = FlowRunner(FakeHelper("Thing", {"thing_empty": ""})) + assert asyncio.run(runner.run(spec)) is False + assert "cannot be checked" in runner.results[-1].detail + + +# ── the workflow runs them on every leg ───────────────────────────────────── + +def test_every_leg_of_the_matrix_runs_the_flows(): + wf = yaml.safe_load((REPO / ".github" / "workflows" / "five-platform-live-qa.yml").read_text()) + legs = 0 + for job in wf["jobs"].values(): + for step in job.get("steps", []): + body = str(step.get("run", "")) + str((step.get("with") or {}).get("script", "")) + body = body.replace("\\\n", " ") # one logical command per line + for call in re.findall(r"testing\.gate\.run_platform[^;\n]*", body): + legs += 1 + assert "--flows testing/flows" in call, f"a leg runs the smoke walk without flows: {call}" + assert legs == 5, f"expected five run_platform legs, found {legs}" + + +# ── run_platform: flows ride the smoke walk, never ahead of it ────────────── + +def test_a_flow_that_does_not_load_stops_the_leg_before_anything_boots(tmp_path, monkeypatch): + from testing.gate import run_platform + bad = tmp_path / "flows" + bad.mkdir() + _flow(bad, GOOD.replace("CSD-900", "CSD-999")) + booted = [] + monkeypatch.setattr(run_platform, "plan_for", lambda a: booted.append(a)) + report = tmp_path / "r.json" + monkeypatch.setattr("sys.argv", ["run_platform", "--platform", "desktop", "--jar", "x.jar", + "--shots", str(tmp_path / "s"), "--report", str(report), + "--flows", str(bad)]) + assert run_platform.main() == 1 + assert booted == [], "a spec error must be found before the app is brought up" + got = yaml.safe_load(report.read_text()) + assert got["steps"][0]["name"] == "flows-load" and not got["ok"] + + +def test_flows_do_not_run_after_a_failed_smoke_walk(tmp_path): + from types import SimpleNamespace + from testing.gate import run_platform + rep = run_platform.Report(platform="desktop") + rep.add("ui-composed", False, "never composed") + run_platform.flows(None, rep, [_spec(tmp_path)], SimpleNamespace(), None) + assert rep.steps[-1].name == "flows" and not rep.steps[-1].ok + assert "not run" in rep.steps[-1].detail + + +# ── navigation: the runner walks to a flow's first screen ─────────────────── + +HOPS = {"Thing": ["circle_x", "tab_y", "nav_thing"]} + + +def _walkable(missing: str = "") -> FakeHelper: + """Lands on Contacts; circle_x -> tab_y -> nav_thing reaches Thing.""" + h = FakeHelper("Contacts", {"circle_x": ""}) + h.leads = { + "circle_x": ("CircleTab", ["circle_x", "tab_y"]), + "tab_y": ("CircleTab", ["circle_x", "tab_y", "nav_thing"]), + "nav_thing": ("Thing", ["thing_list"]), + } + if missing: + for screen, shown in h.leads.values(): + if missing in shown: + shown.remove(missing) + return h + + +def _nav_run(spec, helper, hops=HOPS, flow_only=frozenset()): + return asyncio.run(run_flows.run_one(spec, helper, client_version="0.5.224", + start_timeout=0, hops=hops, flow_only=flow_only)) + + +def test_the_runner_walks_the_derived_hop_to_the_first_screen(tmp_path): + h = _walkable() + out = _nav_run(_spec(tmp_path), h) + assert out.status == run_flows.PASS, out.detail + assert [c for c in h.calls if c.startswith("click")] == [ + "click circle_x", "click tab_y", "click nav_thing"] + + +def test_a_missing_hop_tag_is_cannot_start_and_names_the_tag(tmp_path): + out = _nav_run(_spec(tmp_path), _walkable(missing="tab_y")) + assert out.status == run_flows.CANNOT_START + assert "'tab_y'" in out.detail and "hop 2 of 3" in out.detail + assert "never appeared" in out.detail, "waited for, not blindly clicked" + assert not run_flows.leg_ok([out]) + + +def test_a_screen_with_no_hop_that_is_not_flow_only_cannot_start(tmp_path): + h = _walkable() + out = _nav_run(_spec(tmp_path), h, hops={}) + assert out.status == run_flows.CANNOT_START + assert "no nav hop for Screen.Thing" in out.detail + assert not [c for c in h.calls if c.startswith("click")], "nothing to walk, nothing clicked" + + +def test_a_flow_only_screen_is_waited_for_not_walked_to(tmp_path): + """No hop exists, and that is not a defect: the flow must already be there.""" + h = FakeHelper("Thing", {"thing_list": ""}) + out = _nav_run(_spec(tmp_path), h, hops={}, flow_only={"Thing"}) + assert out.status == run_flows.PASS + elsewhere = FakeHelper("Contacts", {"thing_list": ""}) + out = _nav_run(_spec(tmp_path), elsewhere, hops={}, flow_only={"Thing"}) + assert out.status == run_flows.CANNOT_START and "Thing" in out.detail + assert "no nav hop" not in out.detail, "flow-only is not a missing hop" + + +def test_already_on_the_first_screen_walks_nothing(tmp_path): + h = FakeHelper("Thing", {"thing_list": ""}) + assert _nav_run(_spec(tmp_path), h).status == run_flows.PASS + assert not [c for c in h.calls if c.startswith("click")] + + +def test_the_real_nav_map_reaches_the_seeded_flows_first_screens(): + hops, flow_only = run_flows.nav_hops(has_agent=False) + for spec in run_flows.load_flows([FLOWS]): + start = spec.steps[0].requires.screen + assert start in hops or start in flow_only, f"{spec.flow}: no way to Screen.{start}" + + +def test_navigate_opens_the_single_card_when_a_compact_tab_lists_it(): + """Phones list a one-card tab before opening it (iOS, #97); the runner + opens the only row instead of reporting 'landed on CircleTab'.""" + import asyncio + from testing.gate import run_flows + + class E: + def __init__(self, t): self.test_tag = t + + class H: + def __init__(self): self.screen = "Login"; self.clicked = [] + async def wait_for_element(self, tag, timeout=0): return True + async def click(self, tag, timeout=0): + self.clicked.append(tag) + self.screen = {"tab_people": "CircleTab", "nav_epistemic_contacts": "Contacts"}.get(tag, self.screen) + return True + async def get_screen(self): return self.screen + async def get_elements(self): return [E("nav_epistemic_contacts"), E("tab_people")] + + h = H() + got = asyncio.run(run_flows.navigate(h, "Contacts", ["circle_agent", "tab_people"], + hop_timeout=0.1, arrive_timeout=0.1)) + assert got is None, got + assert h.clicked[-1] == "nav_epistemic_contacts" + + +def test_navigate_prefers_the_target_row_when_a_tab_lists_several(): + import asyncio + from testing.gate import run_flows + + class E: + def __init__(self, t): self.test_tag = t + + class H: + def __init__(self): self.screen = "Login"; self.clicked = [] + async def wait_for_element(self, tag, timeout=0): return True + async def click(self, tag, timeout=0): + self.clicked.append(tag) + self.screen = {"tab_people": "CircleTab", "nav_epistemic_contacts": "Contacts", + "nav_epistemic_community_roster": "CommunityRoster"}.get(tag, self.screen) + return True + async def get_screen(self): return self.screen + async def get_elements(self): + return [E("nav_epistemic_community_roster"), E("nav_epistemic_contacts")] + + h = H() + got = asyncio.run(run_flows.navigate(h, "Contacts", ["circle_agent", "tab_people"], + hop_timeout=0.1, arrive_timeout=0.1)) + assert got is None, got + assert h.clicked[-1] == "nav_epistemic_contacts"