diff --git a/.github/workflows/governance.yml b/.github/workflows/governance.yml index ef30132..331153a 100644 --- a/.github/workflows/governance.yml +++ b/.github/workflows/governance.yml @@ -20,8 +20,12 @@ jobs: strategy: fail-fast: false matrix: - os: [ubuntu-latest, windows-latest] - runs-on: ${{ matrix.os }} + include: + - os: ubuntu-latest + runner: blacksmith-2vcpu-ubuntu-2404 + - os: windows-latest + runner: windows-latest + runs-on: ${{ matrix.runner }} timeout-minutes: 10 steps: - uses: actions/checkout@v4 diff --git a/.github/workflows/update-scribe.yml b/.github/workflows/update-scribe.yml index 6f96222..dfdb0e0 100644 --- a/.github/workflows/update-scribe.yml +++ b/.github/workflows/update-scribe.yml @@ -7,10 +7,11 @@ # Scribe commits in the body, so the bump is reviewed rather than silent. # # Note on CI: GITHUB_TOKEN-created PR runs may require Maintainer approval. -# Approve the pending Governance run, or use Run workflow on the PR branch; -# verify that the successful run covers the current revision before merging. -# Runtime tests also need recorded results; see testing.md. No extra token is -# needed for the manual Governance workflow. +# Approve pending Governance and WFL tests runs, or use Run workflow for both +# workflows on the PR branch. Verify that successful runs cover the current +# revision before merging. +# Link both checks, including the nightly image digest; see testing.md. No extra +# token is needed for manual workflow dispatch. # https://docs.github.com/en/actions/concepts/security/github_token # # To bump by hand instead, run scripts/update-scribe.sh. @@ -135,12 +136,12 @@ jobs: echo echo '| Check or exact command | Result and evidence |' echo '|---|---|' - echo '| `python scripts/run_tests.py --include-scribe` | Not run — this workflow prepares the dependency bump. Record the local and upstream suite results, including failures. |' + echo '| `python3 scripts/run_tests.py --include-scribe` | Pending — this workflow prepares the dependency bump. Link WFL tests results for the current revision, including the nightly image digest and runtime version; record failures. |' echo '| `python -m unittest discover -s tests/tooling -v` | Pending — Governance results are not verified by this workflow. Link results for the current revision. |' echo '| `python scripts/check_repo_hygiene.py` | Pending — Governance results are not verified by this workflow. Link results for the current revision. |' echo '| Affected HTTP/UI rendering journeys | Not run — upstream diff review is needed to identify affected paths. Record setup, expected/actual outcome, and evidence. |' echo - echo 'Approve any pending Governance run, or run Governance manually on this branch. Verify that successful checks cover the current revision before merging.' + echo 'Approve any pending Governance and WFL tests runs, or run both workflows manually on this branch. Verify that successful checks cover the current revision before merging.' echo echo '## Checklist' echo diff --git a/.github/workflows/wfl-tests.yml b/.github/workflows/wfl-tests.yml new file mode 100644 index 0000000..51e3c41 --- /dev/null +++ b/.github/workflows/wfl-tests.yml @@ -0,0 +1,76 @@ +name: WFL tests + +on: + push: + branches: [main] + pull_request: + branches: [main] + workflow_dispatch: + +permissions: + contents: read + +concurrency: + group: wfl-tests-${{ github.ref }} + cancel-in-progress: true + +jobs: + wfl-tests: + name: WFL nightly (Blacksmith) + runs-on: blacksmith-2vcpu-ubuntu-2404 + timeout-minutes: 15 + env: + WFL_IMAGE: bsbyrdwfl/wfl:nightly + TEST_CONTAINER: scriptorium-tests-${{ github.run_id }}-${{ github.run_attempt }} + steps: + - uses: actions/checkout@v4 + with: + persist-credentials: false + submodules: recursive + + - name: Pull latest WFL nightly and record runtime + id: runtime + shell: bash + run: | + docker pull "$WFL_IMAGE" + image="$(docker image inspect --format '{{index .RepoDigests 0}}' "$WFL_IMAGE")" + echo "image=$image" >> "$GITHUB_OUTPUT" + version="$(docker run --rm --network none "$image" --version)" + { + echo '### WFL nightly test environment' + echo + echo "- Image: $image" + echo "- Runtime: $version" + echo "- Scriptorium: $(git rev-parse HEAD)" + echo "- Scribe: $(git -C lib/scribe rev-parse HEAD)" + echo '- Runner: blacksmith-2vcpu-ubuntu-2404 (Linux x64)' + echo '- Command: python3 scripts/run_tests.py --include-scribe' + } | tee -a "$GITHUB_STEP_SUMMARY" + + - name: Run Scriptorium and pinned Scribe suites + shell: bash + env: + RESOLVED_IMAGE: ${{ steps.runtime.outputs.image }} + run: | + # The published image is a minimal runtime with a WFL entrypoint. + # Install Python only in this disposable container. Scribe fixtures + # go to its temporary directory; the checkout stays read-only. + docker run --rm --init --name "$TEST_CONTAINER" \ + --user 0:0 --entrypoint /bin/sh \ + --mount "type=bind,source=$GITHUB_WORKSPACE,target=/work,readonly" \ + --workdir /work "$RESOLVED_IMAGE" -ec ' + apt-get update + apt-get install --yes --no-install-recommends python3 + python3 --version + wfl --version + python3 scripts/run_tests.py --include-scribe + ' + + - name: Clean up test container + if: always() + shell: bash + run: | + # Killing a Docker client on cancellation may leave its container up. + if docker container inspect "$TEST_CONTAINER" >/dev/null 2>&1; then + docker container rm --force "$TEST_CONTAINER" + fi diff --git a/CLAUDE.md b/CLAUDE.md index fe25852..3907df4 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -79,8 +79,11 @@ violate the standard. `python -m unittest discover -s tests/tooling -v` and `python scripts/check_repo_hygiene.py` for repository tooling and hygiene. Python 3.11+ is needed for tooling; `wfl` is needed for application tests. - The Governance workflow runs tooling tests and hygiene on Linux and Windows; - WFL runtime suites currently require recorded local results. See + The Governance workflow runs tooling tests and hygiene on Blacksmith Linux + and GitHub-hosted Windows. The WFL tests workflow runs the application and + pinned Scribe suites on Blacksmith using a freshly pulled `bsbyrdwfl/wfl:nightly` image; + its summary records the resolved image digest, runtime version, and source + revisions. See [testing.md](testing.md) for commands, coverage limits, and merge evidence. - **`data_dir` is an application convention, not a WFL runtime feature.** `main.wfl` reads `.wflcfg` itself at boot and parses the key via diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index dac4564..aa1c9a5 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -121,14 +121,21 @@ write fixtures. Follow [testing.md](testing.md) for HTTP, UI, security, migratio and recovery checks triggered by the change. Prose-only changes need relevant link, command, and hygiene checks; they do not need invented application tests. +The WFL tests workflow runs both application and pinned Scribe suites on +Blacksmith Linux using a freshly pulled `bsbyrdwfl/wfl:nightly` Docker image. +Its summary records the resolved image digest, WFL version, and tested source +revisions. Link the current revision's run as CI runtime evidence; retain any +local regression or boundary evidence needed for the change. Governance checks +repository tooling and hygiene on Blacksmith Linux and GitHub-hosted Windows. + Keep PRs focused and use conventional commit subjects such as `fix:`, `feat:`, `docs:`, `test:`, `refactor:`, and `chore:`. Follow the PR format below. Do not include passwords, session cookies, tokens, user databases, or personal content in logs or fixtures. A review must resolve blocking findings before merge. A bot-created dependency PR needs the same evidence as a human-authored -PR; approve any pending workflow run or run the Governance workflow manually on -the proposed branch, then verify the tested revision and runtime results. +PR; approve any pending workflow runs or run both Governance and WFL tests +manually on the proposed branch, then verify the tested revision and results. GitHub review and branch-protection settings are maintained on the repository host. Adding these documents or workflows does not configure those settings. diff --git a/README.md b/README.md index b9bcfff..6718d27 100644 --- a/README.md +++ b/README.md @@ -217,8 +217,13 @@ python scripts/check_repo_hygiene.py ``` Individual suites still run with `wfl --test TestPrograms/.test.wfl`. -The Governance workflow checks tooling and repository hygiene on Linux and -Windows. Runtime test results must currently be recorded on the PR. See +The [WFL tests workflow](.github/workflows/wfl-tests.yml) runs all five +Scriptorium suites and the pinned Scribe suite on Blacksmith Linux using the +latest `bsbyrdwfl/wfl:nightly` Docker image. It pulls the nightly tag on each run +and records the resolved image digest, runtime version, and tested source +revisions in the job summary. The Governance workflow checks tooling and +repository hygiene on Blacksmith Linux and GitHub-hosted Windows. Link results +for the current revision on the PR. See [testing.md](testing.md) for coverage, commands, and checks for changes to HTTP routes, security, themes, or stored data. @@ -244,8 +249,9 @@ git commit -m "chore(scribe): update lib/scribe" `.github/workflows/update-scribe.yml` does the same thing on a weekly schedule (and on demand via *Run workflow*), opening a PR with the Scribe commits it -picked up. Maintainers must verify Governance checks for the current revision -and record runtime test results before merging; see [testing.md](testing.md). +picked up. Maintainers must verify Governance and WFL tests results for the +current revision before merging; bot-created PRs may need both workflows +dispatched manually on their branch. See [testing.md](testing.md). Delete that file if you would rather bump by hand only. Working on Scribe itself? `lib/scribe` is a normal git checkout — commit and diff --git a/testing.md b/testing.md index 25a1259..6e5f03e 100644 --- a/testing.md +++ b/testing.md @@ -6,7 +6,7 @@ limits of the tooling that exists today. It does not claim that the repository already implements every gate required for a release. - Policy and profile version: 1.0. -- Adopted and reviewed: 2026-09-12. +- Adopted: 2026-09-12; CI profile updated: 2026-09-20. - Test-suite and infrastructure owner: the Maintainer named in [GOVERNANCE.md](GOVERNANCE.md). - Next adoption-gap review: 2026-10-12, or before the next affected behavioral @@ -194,19 +194,33 @@ keyboard behavior, or data integrity. ## CI, merge, and release gates -[.github/workflows/governance.yml](.github/workflows/governance.yml) runs the -repository hygiene and tooling checks. The application suites still require a -local WFL run; there is no pinned-runtime application-test CI job or automated -HTTP/browser journey suite in this repository. The scheduled Scribe updater -proposes dependency changes; it is not runtime test evidence. +[Governance](.github/workflows/governance.yml) runs repository hygiene and tooling +checks on Blacksmith Linux and GitHub-hosted Windows. +[WFL tests](.github/workflows/wfl-tests.yml) runs the five application suites and the pinned Scribe suite via +`python3 scripts/run_tests.py --include-scribe` on +`blacksmith-2vcpu-ubuntu-2404`. Both workflows run for pushes and pull requests to +`main` and support manual dispatch. + +WFL tests pulls `bsbyrdwfl/wfl:nightly` from Docker Hub for every run, resolves +the image digest, and uses that immutable image for that run's runtime checks. +Python is installed in the disposable test container and the source checkout is +mounted read-only. The job summary records the resolved image digest, +`wfl --version`, and tested Scriptorium and Scribe revisions. The nightly tag is +moving: retain the run URL and digest with PR evidence so a later nightly does +not obscure which runtime was tested. The nightly workflow is not a declaration +that every nightly, platform, or production configuration is supported. + +There is no automated HTTP/browser journey suite in this repository. The +scheduled Scribe updater only proposes dependency changes; its successful run +alone is not runtime test evidence. Before merge, required checks MUST pass on the final proposed revision, evidence MUST cover the affected behavior and risk, and blocking reviews MUST be resolved. Recheck after changes that invalidate prior results. Maintainers configure required status checks and review protections on GitHub; repository files alone do not activate host settings. Bot PRs follow the same review and test rules; -approve pending workflow runs or manually dispatch Governance on the proposed -branch and verify that its revision matches the proposal. +approve pending workflow runs or manually dispatch both Governance and WFL tests +on the proposed branch and verify that their revisions match the proposal. Before a production release, the Maintainer MUST identify the immutable Scriptorium and Scribe revisions, runtime version, deployed configuration, and @@ -222,12 +236,18 @@ measurement and passing criteria before making such a claim. ## Adoption gaps and follow-through -The Maintainer owns this dated register as of 2026-09-12. Each item requires an +The Maintainer owns this dated register as of 2026-09-20. Each item requires an owned issue before implementation and a link here when that issue exists. +WFL application-test automation is implemented by the WFL tests workflow under +[issue #13](https://github.com/WebFirstLanguage/Scriptorium/issues/13), owned by +Brad. [PR #14](https://github.com/WebFirstLanguage/Scriptorium/pull/14) retains +the Blacksmith runtime evidence, including an intentional assertion failure +used to verify that failures reach GitHub Actions. Host protection settings +remain a separate adoption item below. + | Gap | Required next step and trigger | |---|---| -| WFL application suites are manual | Pin and provision a runtime in CI; evaluate before the next behavioral PR. Until then attach exact local results to every affected PR. | | No router/HTTP or browser automation | Add real-boundary regression coverage with each affected behavior change; plan coverage of all critical journeys before the next production release. | | No declared compatibility matrix or release-candidate workflow | Define supported runtime/platform/configuration tuples and retain candidate results before the next production release. | | No coverage measurement, performance budgets, or scheduled extended tests | Establish baselines and risk-based targets before claiming those properties; review at the next profile review. |