feat(worklist): the work list is a list of people, and insurance is a… #2324
Workflow file for this run
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| name: CI | |
| on: | |
| # A branch prefix that is NOT in this list gets no push CI, and gets it SILENTLY — the PR simply shows | |
| # no checks, which reads the same as "nothing to run" rather than "nothing ran". `chore/**` and | |
| # `docs/**` were missing, which is how #403 was noticed. The list stays explicit rather than becoming | |
| # "every branch" because the `official-cases` job clones ~34 MB of upstream content and takes up to 20 | |
| # minutes; paying that on every scratch branch is a real cost. If you use a prefix that is not here, | |
| # add it in the same push — for a push event GitHub reads the workflow file from the commit being | |
| # pushed, so it takes effect immediately. | |
| # | |
| # HONEST LIMIT OF THIS FIX: it explains the missing PUSH run and NOT the missing pull_request run. | |
| # `pull_request.branches` filters on the BASE branch, so it is prefix-agnostic and should have fired | |
| # for #403 regardless. PRs #399–#402 each produced both a push run and a pull_request run; #403 | |
| # produced neither, with Actions enabled, no `[skip ci]`, and a mergeable PR. That second cause is | |
| # UNDIAGNOSED. Adding the prefix makes checks appear, which masks the symptom — so if a PR ever again | |
| # shows zero checks, do not assume this line covers it. | |
| push: | |
| branches: [main, "claude/**", "fix/**", "feat/**", "perf/**", "chore/**", "docs/**"] | |
| pull_request: | |
| branches: [main] | |
| workflow_dispatch: | |
| inputs: | |
| e2e_base_url: | |
| description: "E2E target. Allowlisted in .github/scripts/e2e-target-allowed.sh — production is REFUSED, the suite mutates. Default: staging." | |
| type: string | |
| required: false | |
| e2e_profile: | |
| description: "Which Playwright project to run: 'twh' (the existing suite against e2e_base_url) or 'maui' (the Maui pilot project against a stack the runner boots itself as WORKWELL_INSTANCE=maui — no external target, nothing shared is mutated)." | |
| type: choice | |
| options: [twh, maui] | |
| default: twh | |
| required: false | |
| run_scale_maui: | |
| description: "Run the 20,000-patient scale job (measures the pipeline's wall clock; ~5 min, no VSAC needed)." | |
| type: boolean | |
| default: false | |
| required: false | |
| # Weekly rather than nightly: the scale job measures a wall clock, and a number that only ever moves | |
| # when somebody happens to look is worth less than one on a fixed cadence. Sunday 03:00 UTC keeps it | |
| # clear of the deploy windows. | |
| schedule: | |
| - cron: "0 3 * * 0" | |
| # The event name is part of the group so a manual dispatch (the e2e jobs) is not cancelled by the push | |
| # or pull_request run that a `git push` of the same branch starts a second later — measured 2026-09-02: | |
| # three dispatches in a row were "cancelled" before their first step for exactly this reason. | |
| concurrency: | |
| group: ${{ github.workflow }}-${{ github.event_name }}-${{ github.ref }} | |
| cancel-in-progress: true | |
| # Least privilege for all eight jobs. Nothing in this workflow uses GITHUB_TOKEN — no `gh` call, no | |
| # registry push, no commit back — so read access to the code is the whole requirement. Declaring it | |
| # here rather than per job means a job added later inherits the floor instead of silently taking | |
| # whatever the repository default happens to grant. Any job that genuinely needs more overrides this | |
| # block, and a job-level block REPLACES this one rather than adding to it, so an override must | |
| # restate `contents: read`. | |
| permissions: | |
| contents: read | |
| jobs: | |
| deploy-helper: | |
| name: Deploy helper — Shell test | |
| runs-on: ubuntu-latest | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - name: Test bounded MIE manager requests | |
| run: bash .github/scripts/mieweb-api-request.test.sh | |
| # A DELETE that times out at the client may still have been applied by the manager. Aborting | |
| # the deploy there took production down twice (2026-08-30, 2026-09-01), so the read-back that | |
| # resolves the ambiguity is load-bearing — and, like every deploy script, invisible to PR CI | |
| # unless it is tested here. | |
| - name: An ambiguous container DELETE is resolved by reading the manager back | |
| run: bash .github/scripts/mieweb-delete-confirmed.test.sh | |
| # The test above fakes request(), which is the right boundary for the DECISIONS but blind to | |
| # how the real request() behaves. That blind spot let a guard ship that could not run: an | |
| # unconditional `set -e` inside request() re-armed errexit before returning and aborted the | |
| # deploy at the very failure the guard existed to handle (2026-09-01, production). This one | |
| # drives the real request() under the deploy's own `set -euo pipefail` and fakes only curl. | |
| - name: The confirmed delete survives a real transport failure | |
| run: bash .github/scripts/mieweb-delete-confirmed.integration.test.sh | |
| # Deploy workflows only RUN on push to main — i.e. after merge — so a shell syntax error in one is | |
| # invisible to every PR check. PR-9c put an apostrophe inside a single-quoted jq program and turned | |
| # the production deploy step into a parse error; nothing before merge could see it (#356). | |
| - name: Every workflow run-block parses as shell | |
| run: bash .github/scripts/workflow-run-blocks.test.sh | |
| # The E2E allowlist is the only thing standing between a dispatched E2E run and the production | |
| # audit log, and the e2e job itself runs on dispatch only — so nothing on a PR would exercise it. | |
| # Testing it here means a change that weakens the refusal fails on the PR that makes it. | |
| - name: The E2E target allowlist refuses production | |
| run: bash .github/scripts/e2e-target-allowed.test.sh | |
| frontend: | |
| name: Frontend — Lint, Test, Build | |
| runs-on: ubuntu-latest | |
| defaults: | |
| run: | |
| working-directory: frontend | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - uses: pnpm/action-setup@v6 | |
| with: | |
| version: 10.17.1 | |
| run_install: false | |
| - uses: actions/setup-node@v7 | |
| with: | |
| node-version: 24 | |
| cache: pnpm | |
| cache-dependency-path: frontend/pnpm-lock.yaml | |
| - name: Install deps | |
| run: pnpm install --frozen-lockfile | |
| - name: Lint | |
| run: pnpm lint | |
| - name: Unit tests | |
| run: pnpm test | |
| - name: Build | |
| env: | |
| # Build-time only — the real deploys set their own. It pointed at the Fly.io backend | |
| # decommissioned with the Vercel stack (that host no longer resolves), so the built bundle | |
| # baked a dead URL into a check whose whole job is "does this build". | |
| NEXT_PUBLIC_API_BASE_URL: https://twh-api-ts.os.mieweb.org | |
| NEXT_PUBLIC_APP_NAME: WorkWell Measure Studio | |
| NEXT_PUBLIC_DEMO_MODE: "false" | |
| run: pnpm build | |
| backend-ts: | |
| name: Backend-TS — Typecheck & Test | |
| runs-on: ubuntu-latest | |
| # Production runs on the TypeScript backend (backend-ts/), so it must be gated like any other | |
| # code path. Exercises BOTH the SQLite floor (default) and the Postgres ceiling — the postgres | |
| # service + WORKWELL_TEST_PG_URL light up the store-contract suite, which otherwise self-skips. | |
| services: | |
| postgres: | |
| image: postgres:16 | |
| env: | |
| POSTGRES_USER: workwell | |
| POSTGRES_PASSWORD: workwell | |
| POSTGRES_DB: workwell | |
| ports: | |
| - 5432:5432 | |
| options: >- | |
| --health-cmd "pg_isready -U workwell" | |
| --health-interval 10s | |
| --health-timeout 5s | |
| --health-retries 5 | |
| defaults: | |
| run: | |
| working-directory: backend-ts | |
| steps: | |
| # backend-ts consumes @mieweb/cloud from the external/mieweb-cloud submodule (public mieweb/cloud). | |
| - uses: actions/checkout@v7 | |
| with: | |
| submodules: recursive | |
| - uses: pnpm/action-setup@v6 | |
| with: | |
| version: 10.17.1 | |
| run_install: false | |
| - uses: actions/setup-node@v7 | |
| with: | |
| node-version: 24 | |
| cache: pnpm | |
| cache-dependency-path: backend-ts/pnpm-lock.yaml | |
| - name: Install deps | |
| run: pnpm install --frozen-lockfile | |
| - name: Typecheck | |
| run: pnpm typecheck | |
| # Edit a .cql, forget to regenerate, and the tests stay green against the COMMITTED ELM while | |
| # the measure that deploys is the one last compiled — a silent stale-input path, the ADR-040 | |
| # class one layer up (#410). Recompiling here and failing on any difference closes it. Sound | |
| # because the compiler is byte-deterministic: measured for ADR-064, re-measured when this | |
| # landed (two consecutive runs, byte-identical, ~9 s). Three parts are each load-bearing, | |
| # every one mutation-checked before wiring: | |
| # - the `rm` first: the compiler only ever WRITES, so a committed .elm.json the current CQL | |
| # no longer produces (a deleted or version-bumped library) would sit there passing | |
| # forever — the review on #456 found two such orphans already committed. Cleaning first | |
| # turns an orphan into a staged deletion, which fails the diff. | |
| # - the `git add`: a NEW output file is untracked, and `git diff` alone ignores untracked | |
| # files. | |
| # - the paths: they cover ALL THREE outputs the script writes — the ELM, its import index, | |
| # and the bundled translator resources (cql-resources.json), whose first clean-runner run | |
| # caught a real line-ending nondeterminism the local check could not see. | |
| # The vendored CMS artifacts are covered separately by the official-cases job's "reproducible | |
| # from its pin" gate; this covers the AUTHORED measures, which that gate does not see. | |
| - name: The committed ELM is what this CQL produces (#410) | |
| run: | | |
| rm -f src/engine/cql/elm/*.elm.json src/engine/cql/elm/index.ts | |
| pnpm compile-measures | |
| git add -A -- src/engine/cql/elm src/measure/resources/cql-resources.json | |
| git diff --cached --exit-code --stat -- src/engine/cql/elm src/measure/resources/cql-resources.json | |
| # The OpenAPI document must be valid OpenAPI, which `openapi.test.ts` deliberately does NOT check: its | |
| # job is agreement between the document and the running worker, and no amount of that catches a 3.0-ism | |
| # like `nullable` (which 3.1 removed, and which this found on the first run). Two different guards. | |
| # Pinned exactly — `@latest` would make the gate non-reproducible, the same reason the terminology fetch | |
| # is pinned. REDOCLY_TELEMETRY=off is not optional: the CLI otherwise reports environment-variable | |
| # values and the names of the rules that fired. Exits 0 on warnings, non-zero on errors; the five | |
| # expected warnings are explained in src/openapi/spec.ts and are deliberately not ignore-filed. | |
| - name: The OpenAPI document is valid OpenAPI 3.1 | |
| env: | |
| REDOCLY_TELEMETRY: "off" | |
| run: | | |
| node --import tsx scripts/openapi-emit.mjs "$RUNNER_TEMP/openapi.json" | |
| npx --yes @redocly/cli@2.46.1 lint "$RUNNER_TEMP/openapi.json" | |
| - name: Test (SQLite floor + Postgres ceiling) | |
| env: | |
| WORKWELL_TEST_PG_URL: postgres://workwell:workwell@localhost:5432/workwell | |
| run: pnpm test | |
| official-cases: | |
| name: Official MADiE test cases — the eCQM gate | |
| runs-on: ubuntu-latest | |
| timeout-minutes: 20 | |
| # THE RULE (roadmap 2026-07-24 §7.4 PR-6): no measure may enter WORKWELL_OFFICIAL_MEASURES without | |
| # this gate green for it. It is the project's only EXTERNAL ground truth — the measure stewards' | |
| # own expected results — so it is what licenses the claim that we calculate what the measure | |
| # developer intended. | |
| # | |
| # A separate job, not part of `pnpm test`, because it clones ~34MB of official content from | |
| # GitHub at a pinned commit. Keeping it out of the default suite means a developer without network | |
| # access still gets a green local run, while CI always pays the cost. | |
| # | |
| # It also proves the PR-5 vendoring reduction is outcome-neutral: the run compares the upstream | |
| # bundle against our reduced artifact and fails on ANY changed population vector, so this job is | |
| # what would catch a bad `vendor:official` before it reached the deploy image. | |
| defaults: | |
| run: | |
| working-directory: backend-ts | |
| steps: | |
| - uses: actions/checkout@v7 | |
| with: | |
| submodules: recursive | |
| - uses: pnpm/action-setup@v6 | |
| with: | |
| version: 10.17.1 | |
| run_install: false | |
| - uses: actions/setup-node@v7 | |
| with: | |
| node-version: 24 | |
| cache: pnpm | |
| cache-dependency-path: backend-ts/pnpm-lock.yaml | |
| - name: Install deps | |
| run: pnpm install --frozen-lockfile | |
| # The content is pinned to a fixed commit inside the fetch script, so the cache key is stable and | |
| # a hit is always correct. Saves a ~34MB clone on every push. | |
| - name: Cache official content | |
| uses: actions/cache@v6 | |
| with: | |
| path: backend-ts/.official-content | |
| key: official-content-${{ hashFiles('backend-ts/scripts/fetch-official-cases.ps1') }} | |
| # The fetch script is PowerShell (it predates this job); pwsh ships on ubuntu-latest runners. | |
| - name: Fetch official content (pinned commit) | |
| run: pwsh -NoProfile -File scripts/fetch-official-cases.ps1 | |
| # The terminology sidecar is gitignored and fetched at build (ADR-036), so without this step the | |
| # gate silently drops to its weaker fallback — executing our artifact against UPSTREAM's value | |
| # sets rather than the ones the runtime loads, which is the exact gap PR-8a closed. The committed | |
| # report records which mode ran, so the staleness check below is what makes that non-negotiable. | |
| # No separate cache: `vendor:official` reads the bundle out of the `.official-content` checkout | |
| # restored above when it sits at the same pin, so this step downloads nothing. Caching the | |
| # sidecars themselves would have been worse than useless — the script regenerates unconditionally, | |
| # so the cache would be restored and immediately overwritten while still paying two ~17MB pulls. | |
| # `--complete-terminology` re-expands the OIDs upstream capped at 1000 (today one: | |
| # AdvancedIllness, 1000 of 1997, feeding a DENEX in both measures) from VSAC at a pinned release. | |
| # | |
| # It is passed ONLY when the credential is actually present, and that is not merely an | |
| # optimization. GitHub withholds repository secrets from workflows triggered by a FORK pull | |
| # request (Dependabot likewise). Passing the flag unconditionally there would regenerate the | |
| # capped manifests while Git records the completed ones, and the reproducibility check below | |
| # would then fail every outside contributor's PR for a reason that has nothing to do with their | |
| # change. So the step reports whether it ran credentialed, and the check adapts. | |
| - name: Vendor official terminology (the runtime's own expansions, same pin) | |
| id: vendor | |
| env: | |
| WORKWELL_VSAC_API_KEY: ${{ secrets.WORKWELL_VSAC_API_KEY_VENDOR }} | |
| run: | | |
| if [ -n "$WORKWELL_VSAC_API_KEY" ]; then | |
| echo "credentialed=true" >> "$GITHUB_OUTPUT" | |
| COMPLETE=--complete-terminology | |
| else | |
| echo "credentialed=false" >> "$GITHUB_OUTPUT" | |
| COMPLETE= | |
| echo "::notice::No VSAC credential in this context (expected for a fork PR). Vendoring without expansion completion; the reproducibility check will verify bundle bytes only." | |
| fi | |
| # $COMPLETE is deliberately UNQUOTED: it is either one flag or nothing, and quoting it passes | |
| # an empty string that the script's parseArgs rejects as `unknown argument:`. Leave it bare. | |
| pnpm vendor:official --measure CMS122FHIRDiabetesAssessGT9Pct --catalog-id cms122 --strip-elm-annotations --with-tests $COMPLETE | |
| pnpm vendor:official --measure CMS125FHIRBreastCancerScreen --catalog-id cms125 --strip-elm-annotations --with-tests $COMPLETE | |
| # Onboarded 2026-07-30. None of these three has a capped expansion, so $COMPLETE is a no-op for | |
| # them and their manifests are byte-identical credentialed or not — which is why they could be | |
| # vendored without the VSAC key that cms122/cms125 needed (ADR-047). | |
| pnpm vendor:official --measure CMS2FHIRPCSDepScreenAndFollowUp --catalog-id cms2 --strip-elm-annotations --with-tests $COMPLETE | |
| pnpm vendor:official --measure CMS68FHIRDocumentationCurrentMeds --catalog-id cms68 --strip-elm-annotations --with-tests $COMPLETE | |
| pnpm vendor:official --measure CMS951FHIRKidneyHealthEval --catalog-id cms951 --strip-elm-annotations --with-tests $COMPLETE | |
| pnpm vendor:official --measure CMS138FHIRTobaccoScrnCessation --catalog-id cms138 --strip-elm-annotations --with-tests $COMPLETE | |
| pnpm vendor:official --measure CMS130FHIRColorectalCancerScrn --catalog-id cms130 --strip-elm-annotations --with-tests $COMPLETE | |
| pnpm vendor:official --measure CMS165FHIRControllingHighBP --catalog-id cms165 --strip-elm-annotations --with-tests $COMPLETE | |
| # CMS137 (MIPS 305) — the pilot ACO's sixth BLUE measure and the only multi-rate one. Its | |
| # value sets all ship in the upstream bundle, so $COMPLETE is a no-op for it, but it must be | |
| # vendored here all the same: without a sidecar the gate runs it on UPSTREAM terminology, | |
| # which the report-write guard correctly refuses to treat as evidence (ADR-074). | |
| pnpm vendor:official --measure CMS137FHIRSUDTxInitEngagement --catalog-id cms137 --strip-elm-annotations --with-tests $COMPLETE | |
| # Re-vendoring must reproduce the COMMITTED artifact byte for byte. Nothing checked this before: | |
| # the manifest's SHA-256 is written by vendor:official about itself, so it could only ever prove | |
| # self-consistency. This proves the artifact in Git is what the pinned upstream actually produces. | |
| # | |
| # Since PR-9 there is a second way to fail it, and it is a SEQUENCING mistake rather than a bad | |
| # artifact: adding the WORKWELL_VSAC_API_KEY_VENDOR secret without also committing the re-vendored | |
| # manifests means CI now completes the capped expansion while Git still records it as capped. | |
| # The two land together, or this goes red on every unrelated PR — hence the message. | |
| - name: The committed artifact is reproducible from its pin | |
| if: steps.vendor.outputs.credentialed == 'true' | |
| run: | | |
| # `git status --porcelain`, not `git diff`: the vendored MADiE decks are whole FILES, and a | |
| # regenerated-but-UNTRACKED case file is invisible to `git diff`. That would let a deck that | |
| # no longer matches its pin pass this gate silently, which is the one thing it exists to stop. | |
| UNCLEAN=$(git status --porcelain -- measures/official) | |
| if [ -n "$UNCLEAN" ]; then | |
| echo "$UNCLEAN" | |
| git diff -- measures/official || true | |
| echo "::error::measures/official does not match what this pin produces. If WORKWELL_VSAC_API_KEY_VENDOR was just added, re-run 'pnpm vendor:official … --complete-terminology' locally with the key and commit the regenerated manifests in the same change." | |
| exit 1 | |
| fi | |
| # The uncredentialed half of the same check. `bundle.json` is the artifact that decides what gets | |
| # EXECUTED, and completion does not touch it — only `manifest.json` (truncated/completion/the | |
| # terminology digest) and the gitignored sidecar move. So an untrusted PR can still prove the | |
| # committed bundle is what the pinned upstream produces; it just cannot verify the terminology | |
| # half, and says so rather than reporting a pass it did not earn. | |
| - name: The committed bundle is reproducible from its pin (no VSAC credential) | |
| if: steps.vendor.outputs.credentialed != 'true' | |
| run: | | |
| if ! git diff --exit-code measures/official/*/bundle.json; then | |
| echo "::error::the vendored bundle does not match what this pin produces." | |
| exit 1 | |
| fi | |
| echo "::notice::Bundle bytes verified. manifest.json and terminology completion NOT verified — this context has no VSAC credential, so the manifest is regenerated uncompleted and cannot be compared. The credentialed run on merge covers both." | |
| # Every assertion that consults the ARTIFACT'S OWN terminology self-skips without the fetched | |
| # sidecar, and `pnpm test` runs in a job that has no sidecar — so this step is their only gate. | |
| # Any new test that loads a sidecar MUST be added here, or it is permanently skipped in CI while | |
| # reading as covered. That is not hypothetical: PR-8c shipped two such files and had to be told. | |
| - name: Official terminology + corpus tests (need the fetched sidecar) | |
| env: | |
| # `corpus-official-population.test.ts` asserts that at least one measure ACTUALLY RAN when the | |
| # context is credentialed — otherwise a vendor step that quietly stopped producing sidecars | |
| # would make all six tests skip and this job pass. That assertion needs to know whether this | |
| # run was credentialed, and the vendor step above is the only thing that knows. Without this | |
| # line the guard reads the variable, finds nothing, and returns early every time: a guard | |
| # against silent skipping that silently skips. | |
| WORKWELL_REQUIRE_OFFICIAL_TERMINOLOGY: ${{ steps.vendor.outputs.credentialed }} | |
| run: | | |
| pnpm exec node --import tsx --test \ | |
| src/wiring/official-terminology.test.ts \ | |
| src/wiring/corpus-membership.test.ts \ | |
| src/wiring/official-corpus-outcomes.test.ts \ | |
| src/wiring/corpus-official-population.test.ts \ | |
| src/wiring/official-cms165-blood-pressure-identity.test.ts \ | |
| src/standards/literal-diff.test.ts \ | |
| src/engine/ingress/webchart/devdb-official-eval.test.ts \ | |
| src/engine/ingress/webchart/live-official-parity.test.ts \ | |
| src/wiring/official-flip-config.test.ts \ | |
| src/fhir/qrda1-import-official.test.ts \ | |
| scripts/valueset-parity.test.mjs \ | |
| scripts/vendor-official-measure.test.mjs | |
| # `--allow-missing-terminology` ONLY where the credential is genuinely absent (fork PRs, | |
| # Dependabot — GitHub withholds the secret there). Without it, the uncredentialed re-vendor above | |
| # produces a cms138 sidecar that omits the value set upstream does not ship, the deck cannot | |
| # resolve it, and every external contributor's PR goes red for a reason unrelated to their change | |
| # (review of #366). The credentialed run on merge passes NOTHING and therefore skips nothing — | |
| # there, an unresolvable value set means a broken artifact, which is what the gate is for. | |
| - name: Run the official MADiE test-case gate | |
| id: gate | |
| run: | | |
| if [ "${{ steps.vendor.outputs.credentialed }}" = "true" ]; then | |
| pnpm test:official-cases | |
| else | |
| echo "::notice::No VSAC credential here — measures whose upstream bundle omits a value set are SKIPPED rather than failed. The credentialed run on merge covers them." | |
| pnpm test:official-cases --allow-missing-terminology | |
| fi | |
| # The committed report is evidence, so its RESULTS must match what the harness just produced. | |
| # The comparison deliberately ignores the "**Generated:**" line: the harness stamps today's date, | |
| # so a naive `git diff --quiet` would go red the day after every regeneration and train people to | |
| # ignore this job — the worst possible failure mode for a gate. | |
| # Credentialed runs only. An uncredentialed run may skip a measure, which makes it a PARTIAL run — | |
| # the CLI deliberately does not rewrite the committed report there, so comparing against it would | |
| # always differ and would fail every fork PR (review of #366). | |
| - name: Committed evidence report is current (results, not timestamp) | |
| if: steps.vendor.outputs.credentialed == 'true' | |
| working-directory: ${{ github.workspace }} | |
| run: | | |
| report=docs/evidence/OFFICIAL_TESTCASE_REPORT_2026-07.md | |
| strip_date() { grep -v '^\*\*Generated:\*\*' "$1"; } | |
| if ! diff -u <(git show "HEAD:$report" | strip_date /dev/stdin) <(strip_date "$report"); then | |
| echo "::error::$report is stale — run 'pnpm test:official-cases' and commit the result." | |
| exit 1 | |
| fi | |
| echo "Evidence report results are unchanged (timestamp ignored)." | |
| # Only worth uploading once the harness has actually written a report. If the FETCH step failed, | |
| # the file on disk is still the committed copy, and publishing it as "the failing report" is | |
| # actively misleading — it shows green results for a run that never happened. | |
| - name: Upload report on failure | |
| uses: actions/upload-artifact@v7 | |
| if: failure() && steps.gate.outcome != 'skipped' | |
| with: | |
| name: official-testcase-report | |
| path: docs/evidence/OFFICIAL_TESTCASE_REPORT_2026-07.md | |
| wcdb-fhir-shim: | |
| name: WCDB FHIR Shim — Typecheck & Test | |
| runs-on: ubuntu-latest | |
| # The standalone dev/demo shim (ADR-034, #309). Its unit tests run over a STUBBED DB — no | |
| # Docker, no MariaDB — so CI stays container-free; the live parity/acceptance suites remain | |
| # local-only self-skipping tests. Guards the @statement artifact-parsing contract against | |
| # backend-ts's generate:sql output format drifting. | |
| defaults: | |
| run: | |
| working-directory: wcdb-fhir-shim | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - uses: actions/setup-node@v7 | |
| with: | |
| node-version: 24 | |
| cache: npm | |
| cache-dependency-path: wcdb-fhir-shim/package-lock.json | |
| - name: Install deps | |
| run: npm ci | |
| - name: Typecheck | |
| run: npm run typecheck | |
| - name: Test (stubbed DB) | |
| run: npm test | |
| cql-conformance: | |
| name: CQL language conformance — cqframework/cql-tests | |
| runs-on: ubuntu-latest | |
| timeout-minutes: 15 | |
| # V7 (#296 / ADR-060). Runs the 1,823-case CQL language suite through OUR translator | |
| # (@cqframework/cql 4.0.0-beta.1) and OUR engine (cql-execution), and fails on a REGRESSION against | |
| # the committed baseline — never on a bare threshold, because a translator upgrade can trade 30 | |
| # passes for 30 different ones and leave every total identical. | |
| # | |
| # A separate job, not part of `pnpm test`, for the reason the official-cases job records: it clones | |
| # upstream content at a pinned commit, and a developer with no network must still get a green local | |
| # suite. The harness REFUSES to report at all unless it parsed the full corpus (16 files, 1,823 | |
| # cases) — a conformance harness that silently grades a subset publishes a flattering number, which | |
| # is the specific way this job could be worse than useless. | |
| defaults: | |
| run: | |
| working-directory: backend-ts | |
| steps: | |
| - uses: actions/checkout@v7 | |
| with: | |
| submodules: recursive | |
| - uses: pnpm/action-setup@v6 | |
| with: | |
| version: 10.17.1 | |
| run_install: false | |
| - uses: actions/setup-node@v7 | |
| with: | |
| node-version: 24 | |
| cache: pnpm | |
| cache-dependency-path: backend-ts/pnpm-lock.yaml | |
| - name: Install deps | |
| run: pnpm install --frozen-lockfile | |
| # The pin lives inside the fetch script, so keying the cache on that file makes a hit always correct. | |
| - name: Cache cql-tests corpus | |
| uses: actions/cache@v6 | |
| with: | |
| path: backend-ts/.cql-tests | |
| key: cql-tests-${{ hashFiles('backend-ts/scripts/fetch-cql-tests.ps1') }} | |
| - name: Fetch cql-tests (pinned commit) | |
| run: pwsh -NoProfile -File scripts/fetch-cql-tests.ps1 | |
| - name: Run the conformance suite | |
| run: pnpm cql-tests --check | |
| - name: Upload results | |
| uses: actions/upload-artifact@v7 | |
| if: always() | |
| with: | |
| name: cql-conformance-results | |
| path: backend-ts/.cql-tests-results/results.json | |
| packages: | |
| name: Publishable packages — pack and consume outside the workspace | |
| runs-on: ubuntu-latest | |
| timeout-minutes: 15 | |
| # C4 (ADR-063). Everything the workspace supplies for free is invisible to `pnpm test`: whether | |
| # `files` ships what the code needs, whether `publishConfig` really repoints `exports` at `dist/`, | |
| # whether the declared `dependencies` are sufficient, and whether the emitted JS resolves without | |
| # `allowImportingTsExtensions`. Each of those fails silently in-repo and loudly for the first | |
| # integrator, so this job packs real tarballs and installs them into a temp directory that knows | |
| # nothing about this repository. | |
| # | |
| # It runs on every PR rather than only at publish time, because the point is to catch a manifest | |
| # regression when it is introduced — not when someone finally dispatches publish-packages.yml. | |
| # | |
| # A separate job because it needs the network for a plain `npm install` of cql-execution and | |
| # cql-exec-fhir from the registry, and the default suite must stay runnable offline. | |
| defaults: | |
| run: | |
| working-directory: backend-ts | |
| steps: | |
| - uses: actions/checkout@v7 | |
| with: | |
| submodules: recursive | |
| - uses: pnpm/action-setup@v6 | |
| with: | |
| version: 10.17.1 | |
| run_install: false | |
| - uses: actions/setup-node@v7 | |
| with: | |
| node-version: 24 | |
| cache: pnpm | |
| cache-dependency-path: backend-ts/pnpm-lock.yaml | |
| - name: Install deps | |
| run: pnpm install --frozen-lockfile | |
| - name: Pack, install, run and typecheck outside the workspace | |
| run: pnpm verify:publish | |
| e2e: | |
| name: Playwright E2E (manual) | |
| runs-on: ubuntu-latest | |
| if: github.event_name == 'workflow_dispatch' && (inputs.e2e_profile || 'twh') == 'twh' | |
| steps: | |
| - uses: actions/checkout@v7 | |
| - uses: pnpm/action-setup@v6 | |
| with: | |
| version: 10.17.1 | |
| - uses: actions/setup-node@v7 | |
| with: | |
| node-version: 24 | |
| - name: Install e2e dependencies | |
| working-directory: e2e | |
| run: pnpm install | |
| - name: Install Playwright browsers | |
| working-directory: e2e | |
| run: npx playwright install chromium --with-deps | |
| # THE TARGET IS STAGING, NEVER PRODUCTION — this suite MUTATES. `runs.spec.ts` triggers a manual | |
| # run and `case-outreach.spec.ts` POSTs `/actions/outreach`, so against the live TWH stack it would | |
| # create real runs and case actions, and CLAUDE.md's hard rule means every one of those writes an | |
| # `audit_event`. A test suite that pollutes the production audit log is worse than no suite. | |
| # | |
| # It previously pointed at `workwell-measure-studio.vercel.app` — the Vercel stack decommissioned | |
| # when MIE TWH became the sole live one. That host now 404s, so every test burned its full 60s | |
| # timeout plus a retry and the job hung ~14 minutes before being cancelled. It read as E2E coverage | |
| # and could not pass. The preflight below makes that failure mode take seconds and name itself. | |
| # TWO checks, and the ORDER matters. The allowlist runs first and is the one with teeth: a | |
| # reachability check alone approves production, because production is reachable. The description | |
| # on the `e2e_base_url` input is not an enforcement mechanism (review, #407). | |
| - name: Refuse a target that is not on the allowlist | |
| env: | |
| PLAYWRIGHT_BASE_URL: ${{ inputs.e2e_base_url || 'https://twh-staging.os.mieweb.org' }} | |
| run: bash .github/scripts/e2e-target-allowed.sh "${PLAYWRIGHT_BASE_URL}" | |
| - name: Refuse to run against an unreachable target | |
| working-directory: e2e | |
| env: | |
| PLAYWRIGHT_BASE_URL: ${{ inputs.e2e_base_url || 'https://twh-staging.os.mieweb.org' }} | |
| run: | | |
| set -euo pipefail | |
| code="$(curl -s -o /dev/null -w '%{http_code}' --max-time 30 "${PLAYWRIGHT_BASE_URL}" || true)" | |
| echo "${PLAYWRIGHT_BASE_URL} -> HTTP ${code}" | |
| case "${code}" in | |
| 2*|3*) ;; | |
| *) | |
| echo "::error::E2E target ${PLAYWRIGHT_BASE_URL} is not reachable (HTTP ${code}). Deploy staging first, or pass e2e_base_url." | |
| exit 1 | |
| ;; | |
| esac | |
| - name: Run E2E tests | |
| working-directory: e2e | |
| timeout-minutes: 20 | |
| env: | |
| PLAYWRIGHT_BASE_URL: ${{ inputs.e2e_base_url || 'https://twh-staging.os.mieweb.org' }} | |
| run: npx playwright test | |
| - name: Upload Playwright report | |
| uses: actions/upload-artifact@v7 | |
| if: always() | |
| with: | |
| name: playwright-report | |
| path: e2e/playwright-report/ | |
| run-scale-maui: | |
| name: Scale — 20,000-patient run (weekly / manual) | |
| runs-on: ubuntu-latest | |
| # Never on a push or a PR. It takes minutes and measures a wall clock, which on a shared runner is | |
| # noisy enough that gating merges on it would produce flakes rather than findings. The number is | |
| # tracked on a cadence and read by a human. | |
| if: github.event_name == 'schedule' || (github.event_name == 'workflow_dispatch' && inputs.run_scale_maui) | |
| timeout-minutes: 30 | |
| # No VSAC secret and no `.official-content` checkout: this job measures the PIPELINE — corpus | |
| # generation, bundle construction, chunking, the case upserts, batched persistence — with a stub | |
| # engine, and says so in the test's own header. CQL time against the real artifacts is measured in | |
| # `official-cases`, which has the credential. | |
| steps: | |
| - uses: actions/checkout@v7 | |
| with: | |
| submodules: recursive | |
| - uses: pnpm/action-setup@v6 | |
| with: | |
| version: 10.17.1 | |
| - uses: actions/setup-node@v7 | |
| with: | |
| node-version: 24 | |
| - name: Install backend dependencies | |
| working-directory: backend-ts | |
| run: pnpm install --frozen-lockfile | |
| - name: 20,000-patient run | |
| working-directory: backend-ts | |
| env: | |
| WORKWELL_RUN_SCALE_MAUI: "1" | |
| WORKWELL_MAUI_CORPUS_SIZE: "20000" | |
| # The same chunk size the Maui deploy ships, so the measurement describes the deployment | |
| # rather than a default nobody runs. | |
| WORKWELL_RUN_CHUNK_SIZE: "500" | |
| run: pnpm exec node --import tsx --test src/run/run-scale-maui.test.ts | |
| e2e-maui: | |
| name: Playwright E2E — Maui pilot (manual) | |
| runs-on: ubuntu-latest | |
| if: github.event_name == 'workflow_dispatch' && inputs.e2e_profile == 'maui' | |
| timeout-minutes: 45 | |
| # The Maui project runs against a stack THIS RUNNER boots — backend on the SQLite floor with | |
| # WORKWELL_INSTANCE=maui, frontend built in patient terminology with the public-demo affordances | |
| # off — so nothing shared is mutated and the allowlist never comes into play. Browser e2e lives | |
| # here rather than on a developer box because Chromium + two dev servers + a test worker exhaust a | |
| # Windows desktop heap long before they exhaust RAM (measured 2026-09-02). | |
| # | |
| # SINCE 2026-09-08 this job runs the ROUTED configuration — all six of the ACO's measures, the same | |
| # WORKWELL_OFFICIAL_MEASURES the deployed sandbox ships (ADR-078) — by vendoring each measure's | |
| # terminology sidecar with the VSAC credential, exactly as flip-gate.yml does. It previously ran | |
| # authored cms122/cms125 with no VSAC key, which meant the suite's green was about a configuration | |
| # nobody deploys: the four official-only measures sat `official-pending`, and the specs that | |
| # enumerate them failed silently from 2026-09-06 (when #528 added them) until the next manual | |
| # dispatch five days later. A stack that is not the one the pilot runs is not a pilot test. | |
| env: | |
| ROUTED_MEASURES: cms122,cms125,cms2,cms130,cms165,cms137 | |
| steps: | |
| # backend-ts consumes @mieweb/cloud + @mieweb/cli from the external/mieweb-cloud submodule. | |
| - uses: actions/checkout@v7 | |
| with: | |
| submodules: recursive | |
| - uses: pnpm/action-setup@v6 | |
| with: | |
| version: 10.17.1 | |
| - uses: actions/setup-node@v7 | |
| with: | |
| node-version: 24 | |
| - name: Install backend dependencies | |
| working-directory: backend-ts | |
| run: pnpm install --frozen-lockfile | |
| # ---- The credentialed steps run HERE, before the frontend and e2e dependency trees are on disk. | |
| # This job is public-repo CI, and the one exposure that is not covered by "workflow_dispatch needs | |
| # write access" is supply chain: anything in an installed node_modules can read the environment of | |
| # a step it runs in. The VSAC key is therefore scoped to a single step (never job-level `env`), | |
| # and that step executes with the BACKEND tree only — not Chromium, not Next.js, not the e2e tree. | |
| # | |
| # The official artifacts retrieve through their OWN terminology (ADR-036), completed from a | |
| # pinned VSAC release at vendor time. Without it the four official-only measures resolve capped | |
| # expansions, read a zero initial population, and the stack reports them `official-pending`. | |
| - name: Cache official content | |
| uses: actions/cache@v6 | |
| with: | |
| path: backend-ts/.official-content | |
| key: official-content-${{ hashFiles('backend-ts/scripts/fetch-official-cases.ps1') }} | |
| - name: Fetch official content (pinned commit) | |
| working-directory: backend-ts | |
| run: pwsh -NoProfile -File scripts/fetch-official-cases.ps1 | |
| # The completed sidecars are deliberately NOT cached. They are VSAC value-set expansions — | |
| # licensed content, gitignored for that reason — and an Actions cache is restorable by other | |
| # workflow runs in this PUBLIC repo, including a fork pull request whose code we do not control. | |
| # Re-vendoring costs about a minute; putting licensed expansions somewhere fork-authored code can | |
| # read them is not worth a minute. | |
| - name: Vendor each routed measure's terminology, completed from VSAC | |
| working-directory: backend-ts | |
| env: | |
| WORKWELL_VSAC_API_KEY: ${{ secrets.WORKWELL_VSAC_API_KEY_VENDOR }} | |
| run: | | |
| set -o pipefail | |
| if [ -z "$WORKWELL_VSAC_API_KEY" ]; then | |
| echo "::error::No VSAC credential in this context. This job asserts the ROUTED configuration; over capped expansions every official-only measure would read a zero initial population and the suite would call that a pass." | |
| exit 1 | |
| fi | |
| IFS=',' read -ra MEASURES <<< "$ROUTED_MEASURES" | |
| for id in "${MEASURES[@]}"; do | |
| NAME="$(pnpm exec tsx -e "import {officialMeasureName} from './src/standards/official-cases.ts'; process.stdout.write(officialMeasureName('${id}') ?? '')")" | |
| if [ -z "$NAME" ]; then echo "::error::${id} is not an official measure id"; exit 1; fi | |
| pnpm vendor:official --measure "$NAME" --catalog-id "$id" --strip-elm-annotations --complete-terminology | |
| TRUNCATED=$(jq -r '.terminology.truncated | length' "measures/official/${id}/manifest.json") | |
| if [ "$TRUNCATED" != "0" ]; then | |
| echo "::error::${id}'s vendored terminology still reports ${TRUNCATED} TRUNCATED expansion(s) — refusing to run the suite against capped expansions." | |
| exit 1 | |
| fi | |
| done | |
| # Only the gitignored sidecar may be new; the committed artifact must still match its pin. | |
| - name: The committed artifacts are reproducible from their pins | |
| working-directory: backend-ts | |
| run: git diff --exit-code measures/official | |
| # ---- End of the credentialed steps. Everything below runs with no VSAC key in scope. | |
| - name: Install frontend dependencies | |
| working-directory: frontend | |
| run: pnpm install --frozen-lockfile | |
| - name: Cache Chromium | |
| uses: actions/cache@v6 | |
| with: | |
| path: ~/.cache/ms-playwright | |
| key: playwright-chromium-${{ hashFiles('e2e/pnpm-lock.yaml') }} | |
| - name: Install e2e dependencies + Chromium | |
| working-directory: e2e | |
| run: | | |
| pnpm install --frozen-lockfile | |
| npx playwright install chromium --with-deps | |
| - name: Start the backend as Maui | |
| working-directory: backend-ts | |
| env: | |
| WORKWELL_INSTANCE: maui | |
| WORKWELL_AUTH_JWT_SECRET: maui-e2e-ci-secret-key-32-characters-minimum | |
| # The configuration the sandbox deploys (deploy-maui-mieweb.yml), not a reduced stand-in. | |
| WORKWELL_OFFICIAL_MEASURES: ${{ env.ROUTED_MEASURES }} | |
| run: | | |
| nohup pnpm dev > ../backend-maui.log 2>&1 & | |
| for i in $(seq 1 60); do | |
| code="$(curl -s -o /dev/null -w '%{http_code}' --max-time 5 http://localhost:8080/api/version || true)" | |
| [ "$code" = "200" ] && { echo "backend up after $((i*2))s"; exit 0; } | |
| sleep 2 | |
| done | |
| echo "::error::backend did not answer /api/version within 120s"; tail -50 ../backend-maui.log; exit 1 | |
| # The boot line names each measure's state, and this asserts the POSITIVE one: every routed | |
| # measure must read `<id>:official`. | |
| # | |
| # Asserting the absence of `official-pending` instead does almost nothing, which is what the | |
| # first version of this step did. `classifyRunnable` (config/deployment-profile.ts) returns | |
| # `official-pending` for exactly one reason — the id is not named in WORKWELL_OFFICIAL_MEASURES — | |
| # and for cms122/cms125 it cannot return it at all, because both are authored and the `authored` | |
| # branch is reached first. So dropping those two ids from the routing list would print | |
| # `cms122:authored,cms125:authored`, the negative grep would find nothing, and the suite would run | |
| # the pilot's two flagship measures on the AUTHORED engine while the sandbox runs them officially. | |
| # `runs.spec`'s 48 x 6 = 288 holds either way, so the job would be green on the wrong stack. | |
| # | |
| # Note what this can and cannot see: terminology is not part of the classification, so a capped or | |
| # missing sidecar still reads `official` here. The control for THAT is the `truncated` check in the | |
| # vendor step above, which refuses before the backend ever starts. | |
| - name: Every routed measure actually routed | |
| run: | | |
| # The health check above returns as soon as the host listens, and the runnable= line is | |
| # written just after that — so poll for it rather than reading a log that may still be a | |
| # line short. | |
| LINE="" | |
| for _ in $(seq 1 15); do | |
| LINE="$(grep -m1 'runnable=' backend-maui.log || true)" | |
| [ -n "$LINE" ] && break | |
| sleep 2 | |
| done | |
| echo "$LINE" | |
| [ -n "$LINE" ] || { echo "::error::the backend logged no runnable= line"; tail -50 backend-maui.log; exit 1; } | |
| # The line is `[workwell] runnable=cms122:official,cms125:official,...`; drop everything up to | |
| # and including `runnable=` so the FIRST token is a bare `<id>:<kind>` like the rest. | |
| STATES="${LINE##*runnable=}" | |
| IFS=',' read -ra MEASURES <<< "$ROUTED_MEASURES" | |
| for id in "${MEASURES[@]}"; do | |
| # -x on one token per line: an exact match, so `cms2` can never be satisfied by `cms2...` | |
| # and `official` can never be satisfied by `official-pending`. | |
| printf '%s' "$STATES" | tr ',' '\n' | grep -qx "${id}:official" || { | |
| echo "::error::${id} did not boot 'official' — the suite would be testing a stack the pilot does not run" | |
| exit 1 | |
| } | |
| done | |
| - name: Build and start the frontend in patient mode | |
| working-directory: frontend | |
| env: | |
| NEXT_PUBLIC_SUBJECT_TERM: patient | |
| NEXT_PUBLIC_PUBLIC_DEMO: "off" | |
| NEXT_PUBLIC_API_URL: http://localhost:8080 | |
| NEXT_PUBLIC_API_BASE_URL: http://localhost:8080 | |
| NEXT_PUBLIC_APP_NAME: WorkWell Maui | |
| NEXT_PUBLIC_DEMO_MODE: "false" | |
| run: | | |
| pnpm build | |
| nohup pnpm start -p 3000 > ../frontend-maui.log 2>&1 & | |
| for i in $(seq 1 60); do | |
| code="$(curl -s -o /dev/null -w '%{http_code}' --max-time 5 http://localhost:3000/login || true)" | |
| [ "$code" = "200" ] && { echo "frontend up after $((i*2))s"; exit 0; } | |
| sleep 2 | |
| done | |
| echo "::error::frontend did not answer /login within 120s"; tail -50 ../frontend-maui.log; exit 1 | |
| - name: Run the Maui project | |
| working-directory: e2e | |
| timeout-minutes: 25 | |
| env: | |
| PLAYWRIGHT_PROFILE: maui | |
| PLAYWRIGHT_BASE_URL: http://localhost:3000 | |
| PLAYWRIGHT_API_BASE_URL: http://localhost:8080 | |
| # Both Maui projects. `maui-writes` declares `dependencies: ["maui"]`, so the read-only specs | |
| # all finish before the run-triggering one starts, whatever order these are named in. | |
| run: npx playwright test --project=maui --project=maui-writes --reporter=list,html | |
| - name: Upload Playwright report and server logs | |
| uses: actions/upload-artifact@v7 | |
| if: always() | |
| with: | |
| name: playwright-report-maui | |
| path: | | |
| e2e/playwright-report/ | |
| e2e/test-results/ | |
| backend-maui.log | |
| frontend-maui.log |