Skip to content

docs(fabrika): point README readers at the opencode plugin #7303

docs(fabrika): point README readers at the opencode plugin

docs(fabrika): point README readers at the opencode plugin #7303

Workflow file for this run

name: Deploy
# Adapted from #19 (Can Sirin / @cansirin) — alchemy-provisioned CI/CD.
# Fans out over every app under apps/ via the `app` matrix (ADR 0057): phoenix is
# multi-app/multi-worker, each app its own package + alchemy stack + per-app stage,
# reusing the account-global state store + the four CI secrets (no second bootstrap).
on:
push:
branches: [main]
# No `closed`: PR-close teardown moved to its own `pull_request_target` workflow
# (.github/workflows/pr-cleanup.yml, issue #2299) — a `pull_request` `closed` run is
# bound to the PR head ref, which `delete_branch_on_merge` deletes at merge, so GitHub
# cancelled the teardown before it ran. Deploy only needs open/reopen/synchronize.
pull_request:
types: [opened, reopened, synchronize]
concurrency:
group: deploy-${{ github.ref }}
cancel-in-progress: false
# The stage name keys an isolated copy of one app's stack (worker + D1 + DOs).
# Pull requests → `pr-<n>`; pushes to main (the only push trigger) → `prod`.
env:
STAGE: ${{ github.event_name == 'pull_request' && format('pr-{0}', github.event.number) || 'prod' }}
permissions:
contents: read
jobs:
# Job-level change-detection gating ONLY the `deploy` job (issue #2366): a
# docs/canon/pipeline-only PR (`.decisions/`, `.patterns/`, `.glossary/`, `*.md`,
# `.claude/`, `claude-plugins/`, `packages/fabrika-cli/`, …) mints no preview stack
# — no worker, no D1, no DOs, no Flagship app that nothing ever requests.
#
# Why JOB-level and not a workflow-level `paths:`/`paths-ignore:` filter: a
# workflow-level filter would also gate every other job in this workflow, whereas
# only `deploy` should consult it. (Teardown lives in its own workflow now —
# .github/workflows/pr-cleanup.yml, #2299 — so it is unaffected either way.)
#
# THE QUEUE-SAFE INVARIANT (do not invert): the deploy job's RUN-set MUST be a
# SUPERSET of ci.yml's `e2e` RUN-set — deploy skips ONLY a PR where e2e also skips,
# never more. ci.yml's `e2e` job polls deploy.yml's sticky `<!-- preview-deploy -->`
# comment on a 10-minute deadline, so a PR that trips e2e but skips its deploy makes
# e2e poll out → red `ci-required` → wedged PR. So this `deploy` filter's path set is
# the EXACT path set of ci.yml's `changes.e2e` filter (kept identical, cross-referenced
# both ways — see the `e2e:` block in .github/workflows/ci.yml). Equal sets ⇒ deploy's
# path predicate never skips a PR where e2e's does ⇒ superset holds by construction.
# It also FAILS TOWARD DEPLOYING: `push` (prod) and any non-PR event always deploy,
# and any UI change (`apps/**/src/**`, review-design's input — ADR 0165) is in the set,
# so a skipped deploy is provably preview-irrelevant. Deploy's checks are NOT ruleset-
# required (ruleset 17377992 requires only leak-scan / skill-frontmatter / ci-required,
# and deploy.yml has no `merge_group` trigger), so the classic paths-filter/merge-queue
# wedge does not bite here; the e2e-poll coupling above is the real constraint.
changes:
name: detect deploy relevance
runs-on: ubuntu-latest
permissions:
# A job-level block REPLACES the workflow default, so `contents: read` must be
# restated here for the checkout the merge-base step needs. `pull-requests` is no
# longer read by dorny (it runs API-free now) but stays for the paths-filter action's
# own event-payload access.
contents: read
pull-requests: read
outputs:
deploy: ${{ steps.decide.outputs.deploy }}
steps:
# Full history so the merge-base below resolves; the default shallow checkout has no
# common ancestor to compute against.
- uses: actions/checkout@v4.2.2
if: github.event_name == 'pull_request'
with:
fetch-depth: 0
# IDENTICAL CHANGE DETECTION TO ci.yml — same `token`, same `base`, same globs (#3722).
# This job and ci.yml's `changes` job must classify a PR's diff the SAME way or the
# deploy⊇e2e invariant is unenforceable in practice: equal glob lists over DIFFERENT
# changed-file sets still disagree. Two separate defects made them disagree, and both
# are fixed by pinning the pair here and in ci.yml:
# 1. MODE. This step defaulted `token` to `${{ github.token }}` → dorny's API read
# (`pulls.listFiles`, GitHub's three-dot merge-base diff), while ci.yml pins
# `token: ''` → a local `git diff`. Two different readers of "what changed".
# `token: ''` also removes the transient GitHub-API-HTML blip that hard-failed the
# equivalent ci.yml step and redded a defect-free PR (#3244/#3245).
# 2. BASIS. In git mode dorny diffs TWO-DOT from `pull_request.base.sha` — the base
# BRANCH TIP at event time, not the common ancestor — so once main advances past
# the PR's merge ref, main's own drift is reported as this PR's changes.
# Together those wedged PR #3713 (#3722): ci.yml saw 22 phantom `e2e:` hits from main's
# drift → `e2e_required=true`, this job saw the real 4 files → `deploy=false`, and e2e
# polled 10 min for a preview that structurally could never arrive.
- id: mergebase
if: github.event_name == 'pull_request'
run: |
set -uo pipefail
if sha=$(git merge-base "${{ github.event.pull_request.base.sha }}" HEAD 2>/dev/null); then
echo "sha=$sha" >> "$GITHUB_OUTPUT"
echo "::notice::deploy-relevance diff basis = merge base $sha (base.sha ${{ github.event.pull_request.base.sha }})"
else
echo "::warning::could not resolve a merge base — falling back to dorny's base.sha basis (detection stays wide, so this fails TOWARD deploying)"
fi
# Run it only for a PR — `push` (prod) never consults it; the `decide` step below
# forces `deploy=true` for every non-PR event so prod always deploys.
- uses: dorny/paths-filter@v3.0.2
id: filter
if: github.event_name == 'pull_request'
with:
token: ''
base: ${{ steps.mergebase.outputs.sha }}
filters: |
# Deploy-relevant paths = ci.yml's `changes.e2e` filter, EXACTLY (issue #2366).
# Keep this list byte-for-byte in sync with the `e2e:` block in
# .github/workflows/ci.yml — the queue-safe invariant above (deploy-run ⊇
# e2e-run) requires equal path sets, so an edit to one MUST edit the other.
# ENFORCED MECHANICALLY (issue #2372) by `fabrika guard path-filter-guard check`
# (the path-filter-guard.yml job): it fails the build if this `deploy:` set and
# ci.yml's `e2e:` set drift apart — this comment documents it, the guard defends it.
# The guard also compares the two steps' `token`/`base` inputs, because equal
# globs read against different changed-file sets still disagree (#3722).
deploy:
- 'apps/**/src/**'
- 'apps/**/index.html'
- 'apps/**/vite.config.ts'
- 'apps/**/worker/**'
- 'apps/**/alchemy.run.ts'
- 'apps/**/tests/e2e/**'
- 'apps/**/playwright.config.cjs'
- 'apps/**/package.json'
- 'packages/preview-seed/**'
- 'pnpm-lock.yaml'
- '.github/workflows/ci.yml'
- '.github/workflows/deploy.yml'
# Emit a VISIBLE, reported decision (issue #2366 AC): a skipped deploy must not be
# a silently-absent context — the e2e poll + humans see this job's verdict in the
# run log. `push`/non-PR events always deploy (fail toward deploying); on a PR the
# dorny filter decides. This job always completes success, so it never propagates a
# skip to `deploy` (which would skip the prod deploy on `push`).
- id: decide
run: |
set -euo pipefail
if [ "${{ github.event_name }}" != "pull_request" ]; then
echo "deploy=true" >> "$GITHUB_OUTPUT"
echo "::notice::event '${{ github.event_name }}' is not a pull_request → DEPLOY (prod/backstop; fail toward deploying)"
elif [ "${{ steps.filter.outputs.deploy }}" = "true" ]; then
echo "deploy=true" >> "$GITHUB_OUTPUT"
echo "::notice::deploy-relevant paths changed → DEPLOY preview stack (matches ci.yml e2e filter → e2e runs too)"
else
echo "deploy=false" >> "$GITHUB_OUTPUT"
echo "::notice::no deploy-relevant paths changed → SKIP preview deploy (docs/canon/pipeline-only PR). e2e also skips (identical filter); PR-close teardown is a separate workflow (pr-cleanup.yml) and is unaffected."
fi
# Publish the NEGATIVE preview verdict as a first-class artifact (#3722).
#
# ci.yml's `e2e` job cannot `needs:` a job in this workflow, so it synchronizes on the
# sticky `<!-- preview-deploy -->` comment. That seam only ever carried the POSITIVE
# answer, so "no preview is coming" was indistinguishable from "the preview hasn't been
# posted yet" — e2e could only wait out its 10-minute deadline and then red the PR. The
# `changes`/`e2e_required` fix above stops the two workflows from disagreeing in the
# first place; this job makes the failure mode UNREACHABLE rather than merely unlikely,
# by giving the poll a definitive negative to read.
#
# SHA-bound (ADR 0058's stance, applied to a deploy verdict): the marker names the PR head
# it was computed for, so e2e honors it only for the head it is actually running against —
# a stale verdict from an earlier push can never green a later head. The body is REPLACED,
# not upserted, so a previous head's live preview URL is cleared rather than left for e2e
# to probe. This job and the `deploy` matrix are mutually exclusive by construction (their
# `if:`s read the same `needs.changes.outputs.deploy`), so they never race on the comment.
no-preview:
name: report no preview deploy
needs: changes
if: >-
github.event_name == 'pull_request' &&
github.event.action != 'closed' &&
needs.changes.outputs.deploy != 'true'
runs-on: ubuntu-latest
permissions:
pull-requests: write
steps:
- uses: actions/github-script@v7.0.1
env:
HEAD_SHA: ${{ github.event.pull_request.head.sha }}
with:
script: |
const marker = "<!-- preview-deploy -->";
const headSha = process.env.HEAD_SHA;
const body = `${marker}\n### No preview deploy\n`
+ `<!-- preview-deploy:none head:${headSha} -->\n`
+ `- No preview deploy for this PR — its diff touches no deploy-relevant path, `
+ `so no preview stack was minted and \`e2e\` is not applicable. `
+ `<sub>(${headSha.slice(0, 7)})</sub>`;
// Upsert only OUR OWN sticky comment — see ci.yml's `Await preview URL` step for
// why every lookup on this seam is author-constrained (#3740). Without it a third
// party's forged marker is the comment this job edits, so the verdict e2e reads
// sits in an attacker-authored comment they can keep rewriting.
const isTrustedCi = (c) => c.user?.login === "github-actions[bot]" && c.user?.type === "Bot";
const {data: comments} = await github.rest.issues.listComments({
...context.repo, issue_number: context.issue.number, per_page: 100,
});
const existing = comments.find((c) => isTrustedCi(c) && c.body?.includes(marker));
if (existing) {
await github.rest.issues.updateComment({...context.repo, comment_id: existing.id, body});
} else {
await github.rest.issues.createComment({...context.repo, issue_number: context.issue.number, body});
}
core.info(`published no-preview verdict for head ${headSha}`);
# Single source of the per-app roster (ADR 0057, #1434), declared ONCE in
# .github/app-roster.json and consumed via `fromJSON(needs.app-roster.outputs.matrix)`
# by BOTH this workflow's `deploy` AND pr-cleanup.yml's `cleanup`, so an app added to
# deploy can't be silently missed on teardown — the deploy/cleanup-drift that leaks a
# worker + D1 + DOs on PR close, the exact class the teardown-verify step (#813)
# exists to catch. GitHub Actions has no YAML-anchor reuse across jobs/workflows, so a
# roster job emitting the matrix JSON from the shared file is the single-source
# mechanism; adding an app is one declaration edit in app-roster.json, not a shotgun.
app-roster:
name: resolve app roster
runs-on: ubuntu-latest
outputs:
matrix: ${{ steps.roster.outputs.matrix }}
steps:
# Single source of the app roster (.github/app-roster.json), shared with the
# pr-cleanup.yml `app-roster` job so deploy and teardown can never drift apart —
# the deploy/cleanup-drift stage leak #1434 killed. `app` is the apps/ dir;
# `pnpm-name` its workspace package; `needs-auth` whether the app's worker binds
# BETTER_AUTH_SECRET (@kampus/web reads it `Effect.orDie` in config.ts, so its
# deploy/destroy passes the secret; a future auth-less app sets `needs-auth:
# false` so its deploy never requires it — ADR 0057).
- uses: actions/checkout@v4.2.2
- id: roster
run: |
set -euo pipefail
echo "matrix=$(jq -c . .github/app-roster.json)" >> "$GITHUB_OUTPUT"
deploy:
# Defense-in-depth: never run a secret-bearing deploy for a fork PR. A
# same-repo (non-fork) PR can only be opened by someone with push access, so
# it's inherently trusted to run with secrets; a fork PR is the real
# secret-leak risk. The repo's fork-PR approval policy already gates forks
# (all_external_contributors → manual "Approve and run"), but encoding the
# fork gate here makes it explicit and survives any settings drift. `push`
# (prod, main-only) always proceeds; PRs run only when the head repo IS this
# repo (not a fork). This replaces the old `author_association` gate, whose
# webhook-payload value lagged org-membership visibility and skipped the
# deploy on maintainer same-repo PRs (issue #293).
#
# Dependabot guard (issue #1901): a `dependabot[bot]`-authored PR is a
# SAME-REPO PR (its `dependabot/*` branch lives in this repo, not a fork), so
# it passes the fork gate above — yet GitHub runs Dependabot-triggered
# workflows against a SEPARATE `dependabot` secrets store, so the four CI
# secrets (CLOUDFLARE_*/ALCHEMY_PASSWORD/BETTER_AUTH_SECRET) resolve EMPTY on
# its runs and `alchemy deploy` dies with AuthError. See GitHub's docs,
# "Automating Dependabot with GitHub Actions" → "Accessing secrets": a
# Dependabot-triggered `pull_request` run cannot read regular Actions secrets;
# they must be set as `dependabot`-scoped secrets to be visible. Populating
# that separate store would hand the deploy creds to auto-opened dep-bump runs
# (a wider blast radius wanting its own decision), so instead SKIP the deploy
# for Dependabot. `github.actor` is `dependabot[bot]` on those runs (verified
# on run 28685219958). This is additive — the fork gate is preserved, and the
# skip is scoped to `dependabot[bot]`, so it opens no path for a fork/untrusted
# PR to skip the deploy. A skipped matrix job reports neutral (not red) to the
# merge queue's ALLGREEN policy (ruleset 17377992), so the Dependabot PR greens
# and can merge instead of wedging on a deploy that structurally can't
# authenticate; the `push`-to-`main` (prod) path is `github.actor`-independent
# and unchanged.
#
# Deploy-relevance gate (issue #2366): only mint a preview stack when the diff
# touches a deploy-relevant path (== ci.yml's `e2e` filter — see the `changes` job
# above). `push` (prod) forces `changes.outputs.deploy == 'true'`, so this term never
# gates the prod deploy; a docs/canon/pipeline-only PR skips it. `changes` always
# completes success, so consuming its output here never skip-propagates onto prod.
if: >-
github.event.action != 'closed' &&
github.actor != 'dependabot[bot]' &&
needs.changes.outputs.deploy == 'true' &&
(github.event_name == 'push' ||
github.event.pull_request.head.repo.full_name == github.repository)
needs: [app-roster, changes]
runs-on: ubuntu-latest
permissions:
contents: read
pull-requests: write # post the preview-URL comment on PRs
strategy:
# Each app is an independent worker + stack (ADR 0057) — one app's deploy
# failing must not cancel the other's. `fail-fast: false` keeps both running.
fail-fast: false
# The roster is single-sourced in .github/app-roster.json (#1434) and shared with
# pr-cleanup.yml, so deploy and teardown can never drift onto different app sets.
matrix: ${{ fromJSON(needs.app-roster.outputs.matrix) }}
steps:
- uses: actions/checkout@v4.2.2
with:
# On a `pull_request` event we deliberately keep checkout's default
# `refs/remotes/pull/N/merge` (the PR merged into base `main`) so the
# deploy exercises the merged tree, surfacing main-drift/conflicts the
# bare head SHA wouldn't. `fetch-depth: 0` makes that merge commit's
# PARENTS reachable, which the stale-merge-ref guard below needs to
# read `HEAD^2` and to recompute the merge locally if it's stale.
fetch-depth: 0
# Reject (and repair) a stale pull/N/merge ref before building (issue #917).
# In the first minutes after a push — exactly when CI fires — GitHub's
# computed `pull/N/merge` can still be the merge of the PR's PRIOR head into
# main, so the default checkout builds the WRONG commit and false-FAILs code
# that is green at head (confirmed on PR #914: head 0a7a463 was green, but
# `pull/914/merge` was the merge of the already-superseded bca1889).
#
# The merge commit's second parent IS the PR head it was computed against, so
# `HEAD^2 == github.event.pull_request.head.sha` is the exact staleness test.
# On a match we build the merge ref unchanged (full main-drift coverage kept).
# On a mismatch the ref is stale: instead of failing the run (which would only
# self-correct on a re-run), we recompute the SAME merged-into-main tree
# LOCALLY from the event's known head SHA + the live base tip — preserving
# merged-into-main semantics while rejecting the stale build. `push`-to-main
# (prod) has no merge ref, so this step is `pull_request`-only and the prod
# path is untouched.
- name: Guard against a stale pull/N/merge ref (PR only)
if: github.event_name == 'pull_request'
env:
PR_HEAD_SHA: ${{ github.event.pull_request.head.sha }}
PR_BASE_REF: ${{ github.event.pull_request.base.ref }}
run: |
set -euo pipefail
second_parent="$(git rev-parse HEAD^2)"
if [ "$second_parent" = "$PR_HEAD_SHA" ]; then
echo "merge ref is fresh: HEAD^2 ($second_parent) == event head SHA ($PR_HEAD_SHA)"
exit 0
fi
echo "::warning::stale pull/N/merge — HEAD^2 ($second_parent) != event head SHA ($PR_HEAD_SHA); recomputing the merge locally"
# Fetch the exact head the event fired for and the live base tip, then
# rebuild the merge deterministically. `--no-edit` keeps it non-interactive;
# a real merge conflict still fails the step (that conflict IS the main-drift
# signal the merge ref exists to surface — preserved, not papered over).
git fetch --no-tags --depth=1 origin "$PR_HEAD_SHA"
git fetch --no-tags --depth=1 origin "$PR_BASE_REF"
git -c user.name=ci -c user.email=ci@local checkout -B deploy-merge "origin/$PR_BASE_REF"
git -c user.name=ci -c user.email=ci@local merge --no-edit "$PR_HEAD_SHA"
rebuilt_second_parent="$(git rev-parse HEAD^2)"
if [ "$rebuilt_second_parent" != "$PR_HEAD_SHA" ]; then
echo "::error::failed to rebuild merge against event head SHA ($PR_HEAD_SHA); refusing to deploy the wrong commit"
exit 1
fi
echo "rebuilt merge of head $PR_HEAD_SHA into base $PR_BASE_REF"
- uses: pnpm/action-setup@v4.1.0
- uses: actions/setup-node@v4.4.0
with:
node-version-file: package.json
cache: pnpm
- run: pnpm install --frozen-lockfile
# Map the alchemy stage name → the deploy ENVIRONMENT class (ADR 0088) via the
# single owner of that map, apps/web/worker/environment.ts (`environmentForStage`),
# so the `prod`→`production` spelling lives in ONE place instead of being inlined in
# a YAML expression here (#1433). node strips the module's types; it has no runtime
# deps, so this resolves against the checked-out source alone. Three classes: `prod`
# → `production`; every per-PR `pr-<n>` preview → `preview`, which trusts its OWN
# `*.kampusinfra.workers.dev` served origin for better-auth (fixes #704 — a preview
# labelled `development` hardcoded localhost origins and rejected real browser
# sign-up as "Invalid origin"). `development` is local `alchemy dev` only (set in
# `.env`), never a deployed stage. Sets ENVIRONMENT in $GITHUB_ENV for the deploy step.
- name: Resolve deploy ENVIRONMENT from stage
run: |
set -euo pipefail
ENVIRONMENT="$(node --eval "import('./apps/web/worker/environment.ts').then((m) => process.stdout.write(m.environmentForStage(process.env.STAGE)))")"
echo "stage '$STAGE' → ENVIRONMENT '$ENVIRONMENT'"
echo "ENVIRONMENT=$ENVIRONMENT" >> "$GITHUB_ENV"
# The SPA is a plain Vite build; the worker uploads `dist/client` via its
# `assets` prop. `alchemy deploy` does not drive Vite for phoenix's shape.
- name: Build SPA (${{ matrix.app }})
env:
# Bake the Sentry DSN into the bundle at build time (ADR 0118 — a public
# client-side value, so a repo variable, not a secret). Production stages
# only: an empty DSN leaves the SPA wiring inert (sentry.ts `sentryEnabled`
# gate), so per-PR previews neither init Sentry nor burn the free-tier quota.
VITE_SENTRY_DSN: ${{ env.ENVIRONMENT == 'production' && vars.VITE_SENTRY_DSN || '' }}
# Snapshot-hydration containment flag (#2319/#2333) — same build-time repo-variable
# gate as the Sentry DSN above: prod reads the variable (its value IS the release
# state), previews build flag-dark (empty ≠ "on"). The variable's absence = flag off.
VITE_FEED_SNAPSHOT: ${{ env.ENVIRONMENT == 'production' && vars.VITE_FEED_SNAPSHOT || '' }}
run: pnpm --filter ${{ matrix.pnpm-name }} build
# `exec alchemy` forwards `--stage`/`--yes` to the alchemy CLI. A bare
# `pnpm --filter <app> deploy --stage …` makes *pnpm* eat the flags
# ("Unknown options: 'stage', 'yes'"). `--yes` skips the interactive plan
# approval (CI would otherwise hang); `--stage` selects the isolated copy.
# Creds come from the secrets `infra/ci-credentials/github.ts` provisions
# (run via `pnpm --filter @kampus/infra exec alchemy deploy github.ts`); BETTER_AUTH_SECRET
# is passed only for apps whose worker binds it (matrix.needs-auth).
- name: Deploy ${{ matrix.app }} (stage ${{ env.STAGE }})
id: deploy
run: |
set -o pipefail
pnpm --filter ${{ matrix.pnpm-name }} exec alchemy deploy --stage "$STAGE" --yes | tee deploy.log
# alchemy has no structured URL output, so scrape the worker URL it
# prints (`url: 'https://…workers.dev'`). `set -o pipefail` keeps the
# deploy's exit code through the `tee` so a failed deploy still fails.
echo "url=$(grep -oE 'https://[A-Za-z0-9.-]+\.workers\.dev' deploy.log | tail -1)" >> "$GITHUB_OUTPUT"
env:
CI: "true"
# ENVIRONMENT comes from the "Resolve deploy ENVIRONMENT from stage" step above
# (via $GITHUB_ENV) — owned by apps/web/worker/environment.ts, not inlined here (#1433).
# CI-secret roster: the four secrets below are enumerated canonically in
# infra/ci-credentials/github.ts (the provisioner that mints/pushes them);
# keep this set in sync with that roster (#1432).
CLOUDFLARE_API_TOKEN: ${{ secrets.CLOUDFLARE_API_TOKEN }}
CLOUDFLARE_ACCOUNT_ID: ${{ secrets.CLOUDFLARE_ACCOUNT_ID }}
ALCHEMY_PASSWORD: ${{ secrets.ALCHEMY_PASSWORD }}
# The session-signing secret, bound only by apps with auth. `config.ts`
# reads it via `Config.redacted(ENV_BINDINGS.betterAuthSecret)` → a `secret_text`
# binding the worker reads at runtime; the value must be present at deploy
# time. A future auth-less app sets `needs-auth: false` and never reads it.
BETTER_AUTH_SECRET: ${{ matrix.needs-auth && secrets.BETTER_AUTH_SECRET || '' }}
# Sentry DSN for the WORKER tier (ADR 0118, #1502). index.ts adds the
# `SENTRY_DSN` secret_text binding only when this is present at deploy; absent
# ⇒ no binding ⇒ the worker Sentry path stays inert. Same public DSN as the SPA
# (the `VITE_SENTRY_DSN` repo variable), production stages only.
SENTRY_DSN: ${{ env.ENVIRONMENT == 'production' && vars.VITE_SENTRY_DSN || '' }}
# Make "prod serves" — not "apply exited 0" — the prod deploy's success signal
# (issue #1436). `alchemy deploy` exiting 0 proves only that the apply reconciled;
# a cleanly-applied stack can still serve a broken worker (bad binding, failed
# migration side-effect, un-provisioned TLS — the #983 class) and go live green.
# PR previews are already behaviorally verified (ci.yml e2e hits the preview URL),
# so prod — the one environment users actually hit — had the weakest gate. This is
# the prod-side analogue of the cleanup teardown-verify (#813, now in pr-cleanup.yml): a real
# post-action behavioral gate that fails loudly rather than pass silently.
#
# Prod-only by construction: gated on `github.event_name == 'push'` (the sole push
# trigger is main → STAGE=prod, per the env block above), so PR `pr-<n>` previews —
# already covered by ci.yml e2e — never run it. Bounded retry absorbs Cloudflare's
# post-deploy propagation window (a just-deployed worker can 5xx/DNS-miss for a few
# seconds), mirroring the teardown-verify's retry loop; on exhaustion it FAILS the
# job (no false-green) — an unreachable prod, a non-200, or a body whose `status`
# is not "ok" (the health handler returns `{status:"ok",…}`, see
# apps/web/worker/http/health.ts) all red the deploy run.
#
# The URL-presence check lives INSIDE the step body, NOT in the `if:` guard (issue
# #1617). The guard is `github.event_name == 'push'` ALONE: an empty `outputs.url`
# after a "successful" deploy — an alchemy log-scrape miss or a genuinely URL-less
# deploy — must RED the job, not skip it green. A `&& steps.deploy.outputs.url != ''`
# clause in the `if:` would make an empty URL SKIP the step, which GitHub records as
# a passed job: a fail-OPEN corner in the very gate meant to be fail-closed against a
# broken prod (ADR 0092, gates fail closed on zero scope). So on a prod push the step
# always runs, and an empty URL exits non-zero with `::error::` — "no URL to check"
# is itself a deploy failure worth surfacing.
- name: Verify prod serves — GET /api/health (${{ matrix.app }})
if: github.event_name == 'push'
env:
PROD_URL: ${{ steps.deploy.outputs.url }}
run: |
set -uo pipefail
# Fail closed on an empty URL (issue #1617): `alchemy deploy` exited 0 but the
# scrape yielded no worker URL, so there is nothing to probe — red the deploy
# rather than pass silently, same fail-closed posture as the retry exhaustion below.
if [ -z "$PROD_URL" ]; then
echo "::error::prod health gate FAILED (issue #1617) — alchemy deploy exited 0 but emitted no scrapeable worker URL."
echo "An empty URL after a successful deploy is itself a deploy failure. Failing so it surfaces instead of skipping green (ADR 0092)."
exit 1
fi
healthy=""
last=""
# Retry across CF's propagation window (same shape as the teardown-verify loop):
# increasing back-off, break the moment prod answers healthy.
for attempt in 1 2 3 4 5 6; do
code="$(curl -s -o /tmp/health.json -w '%{http_code}' --max-time 20 \
"${PROD_URL}/api/health" || echo 000)"
status="$(jq -r '.status // ""' /tmp/health.json 2>/dev/null || echo "")"
last="http=${code} status=${status}"
if [ "$code" = "200" ] && [ "$status" = "ok" ]; then
healthy=yes
break
fi
echo "attempt ${attempt}: prod not healthy yet (${last}); retrying"
sleep $((attempt * 5))
done
if [ -z "$healthy" ]; then
echo "::error::prod health gate FAILED (issue #1436) — ${PROD_URL}/api/health did not serve a healthy 200 (last: ${last})."
echo "alchemy deploy exited 0 but prod does not serve. Failing so the broken deploy surfaces instead of going live green."
cat /tmp/health.json 2>/dev/null || true
exit 1
fi
echo "prod health verified: ${PROD_URL}/api/health → 200 status=ok"
# Resolve the deployed web stage's D1 UUID so the e2e job (ci.yml, #522) can
# seed it before running the unauth read specs. The seed needs the physical
# database id; alchemy's per-stage D1 carries a random 16-char suffix
# (`createPhysicalName`), so the id is NOT reconstructable from the stage
# name alone and must be looked up here, where the creds + stage are in hand.
# Alchemy names the D1 `${stack}-${id}-${stage}-${suffix}` then sanitizes it
# (`_`→`-`), so the id "phoenix_db" becomes "phoenix-db" in the physical name:
# `phoenix-phoenix-db-<stage>-…` (alchemy.run.ts stack "phoenix"; resources.ts
# D1 id "phoenix_db"). Do NOT "fix" the prefix back to `phoenix_db` — the CF
# `/d1/database?name=` filter is a substring match, so this prefix uniquely
# selects this stage's database. web-only (`@kampus/web` is the app the e2e
# specs target) and PR-only (no preview seed on prod).
- name: Resolve web preview D1 id (${{ matrix.app }})
id: d1
if: github.event_name == 'pull_request' && matrix.app == 'web'
env:
CLOUDFLARE_API_TOKEN: ${{ secrets.CLOUDFLARE_API_TOKEN }}
CLOUDFLARE_ACCOUNT_ID: ${{ secrets.CLOUDFLARE_ACCOUNT_ID }}
DB_NAME_PREFIX: phoenix-phoenix-db-${{ env.STAGE }}-
run: |
set -euo pipefail
resp="$(curl -sf -G \
"https://api.cloudflare.com/client/v4/accounts/${CLOUDFLARE_ACCOUNT_ID}/d1/database" \
-H "Authorization: Bearer ${CLOUDFLARE_API_TOKEN}" \
--data-urlencode "name=${DB_NAME_PREFIX}")"
# The prefix-filtered list should hold exactly one match for this stage.
uuid="$(echo "$resp" | jq -r --arg p "$DB_NAME_PREFIX" \
'[.result[] | select(.name | startswith($p))] | (.[0].uuid // "")')"
if [ -z "$uuid" ] || [ "$uuid" = "null" ]; then
echo "::error::could not resolve D1 uuid for prefix '${DB_NAME_PREFIX}'"; exit 1
fi
echo "uuid=$uuid" >> "$GITHUB_OUTPUT"
# Force per-PR preview flags ON from `preview-flag:<key>` PR labels — deploy-time
# serving write (issue #2951; mechanism A2, decided in investigation #2946). A label
# `preview-flag:<key>` forces that flag key ON in THIS PR's own `pr-<n>` Flagship app,
# so the release-blocking flag-ON journey e2e (epic #2926) and a human clicking the
# preview link both see flag-gated UI in its ON state — previews are otherwise
# flag-dark (each stage's Flagship app mints at IaC defaults, all off). The write is
# CI-credentialed and direct-to-store via anka-ops' `setServing` seam (ADR 0134),
# with ZERO runtime surface in the worker bundle — the preview-seed idiom, the
# deliberate counter-shape of the deleted fail-open seeder (CLAUDE.md "Sözlük seed").
#
# PRODUCTION HARD-REFUSE — the load-bearing invariant this whole step exists to keep:
# forcing must NEVER reach the prod Flagship app. Two independent gates. (1) The step
# is `pull_request`-only, and prod deploys only on `push` (STAGE=prod), so it never
# runs on a prod deploy. (2) Inside the step, re-resolve `environmentForStage($STAGE)`
# from its single owner (apps/web/worker/environment.ts — only `prod` maps to
# `production`) and `exit 1` BEFORE any write when it is `production`. The in-step
# recompute is deliberately self-contained (not the earlier step's $ENVIRONMENT) so
# the guard can't be defeated by an upstream edit; it fails closed (ADR 0092) — the
# refusal aborts the step, never downgrades to a silent no-op.
#
# web-only: the Flagship app (`phoenix_flags`) lives in the web stack
# (apps/web/worker/features/flagship/resources.ts). Idempotent: anka-ops `flag open` is
# a converging upsert — a re-run on a later push writes nothing when already ON.
- name: Force preview flags ON from PR labels (${{ matrix.app }})
id: force-flags
if: github.event_name == 'pull_request' && matrix.app == 'web'
env:
# The store write is Cloudflare-credentialed, NOT GITHUB_TOKEN. Least privilege:
# the label read below uses the default job token, already scoped by the job's
# `permissions:` (contents: read, pull-requests: write — GitHub Actions scopes
# permissions per JOB, not per step); this step adds no GitHub token scope.
CLOUDFLARE_API_TOKEN: ${{ secrets.CLOUDFLARE_API_TOKEN }}
CLOUDFLARE_ACCOUNT_ID: ${{ secrets.CLOUDFLARE_ACCOUNT_ID }}
GH_TOKEN: ${{ github.token }}
PR_NUMBER: ${{ github.event.number }}
run: |
set -euo pipefail
# GATE (structural production refuse): recompute the deploy class from the single
# owner of the stage→ENVIRONMENT map and abort before any write if it is
# production — belt-and-suspenders over the `pull_request`-only `if:` above.
ENVIRONMENT_FOR_STAGE="$(node --eval "import('./apps/web/worker/environment.ts').then((m) => process.stdout.write(m.environmentForStage(process.env.STAGE)))")"
if [ "$ENVIRONMENT_FOR_STAGE" = "production" ]; then
echo "::error::preview flag-forcing REFUSED — stage '$STAGE' resolves to ENVIRONMENT 'production'. This path must never touch the prod Flagship app (issue #2951; fail-closed, ADR 0092)."
exit 1
fi
echo "stage '$STAGE' → ENVIRONMENT '$ENVIRONMENT_FOR_STAGE' (non-prod) — preview flag-forcing permitted"
# Read the PR's `preview-flag:<key>` labels fresh from the API (so a label added
# after this deploy triggered is still honored on this run), stripping the prefix
# to the bare flag key.
keys="$(gh api "repos/${{ github.repository }}/issues/${PR_NUMBER}/labels" \
--jq '.[].name | select(startswith("preview-flag:")) | sub("^preview-flag:"; "")')"
if [ -z "$keys" ]; then
echo "no preview-flag:<key> labels on PR #${PR_NUMBER} — nothing to force"
echo "forced=" >> "$GITHUB_OUTPUT"
exit 0
fi
# Force each labeled key ON in THIS stage's Flagship app. `flag open <key>
# --env <stage> --execute` resolves the app whose decoded stage == $STAGE
# (anka-ops decodeEnv/findAppForEnv) and serves on@100% via the no-match split.
# A `flag open` failure (unknown key, no app for the stage) aborts under `set -e`
# — fail-loud, never a silently-skipped force.
forced=""
while IFS= read -r key; do
[ -z "$key" ] && continue
echo "forcing flag '$key' ON in Flagship env '$STAGE'"
node packages/anka-ops/src/bin.ts flag open "$key" --env "$STAGE" --execute
forced="${forced:+$forced,}$key"
done <<< "$keys"
echo "forced=$forced" >> "$GITHUB_OUTPUT"
echo "forced preview flags: ${forced:-none}"
# Post (or update) a sticky comment carrying EVERY app's preview URL on PRs.
# The outer `<!-- preview-deploy -->` marker makes the comment sticky; a
# per-app `<!-- preview-deploy:<app> -->` line lets each matrix job upsert
# only its own URL into the shared comment. Parallel matrix jobs both edit
# the one comment, so the upsert is an optimistic read-modify-write retried
# on conflict (ETag 409 / lost update). Skipped on pushes to main (`prod`).
- name: Comment preview URL (${{ matrix.app }})
if: github.event_name == 'pull_request' && steps.deploy.outputs.url != ''
uses: actions/github-script@v7.0.1
env:
APP: ${{ matrix.app }}
PREVIEW_URL: ${{ steps.deploy.outputs.url }}
STAGE_NAME: ${{ env.STAGE }}
# web-only: the e2e job (ci.yml, #522) parses this token to seed the D1.
DB_ID: ${{ steps.d1.outputs.uuid }}
# web-only: the forced preview-flag key set (issue #2951) the e2e run reads as
# E2E_FORCED_FLAGS, so darkship OFF-assertion specs can skip/invert against a
# forced-ON preview. Empty when no `preview-flag:<key>` labels forced anything.
FORCED_FLAGS: ${{ steps.force-flags.outputs.forced }}
# Empty on a push to main; the script falls back to `context.sha` there.
PR_HEAD_SHA: ${{ github.event.pull_request.head.sha }}
with:
script: |
const marker = "<!-- preview-deploy -->";
const app = process.env.APP;
const appMark = `<!-- preview-deploy:${app} -->`;
// Under `pull_request` the checkout is refs/pull/N/merge, so `context.sha` names the
// merge commit, not the branch head. `review-ui render` compares this stamp to the PR's
// live head, and a merge sha never matches it (#6507).
const sha = (process.env.PR_HEAD_SHA || context.sha).slice(0, 7);
// Hidden, machine-parseable tokens the e2e job reads. Only web resolves a D1 id
// (#522) and a forced-flag set (#2951); other apps omit both. Kept in HTML
// comments so they don't clutter the human-facing line.
const dbId = process.env.DB_ID || "";
const dbToken = dbId ? ` <!-- d1:${dbId} -->` : "";
const forcedFlags = process.env.FORCED_FLAGS || "";
const flagsToken = forcedFlags ? ` <!-- preview-flags:${forcedFlags} -->` : "";
const appLine = `${appMark}\n`
+ `- **${app}** — Stage \`${process.env.STAGE_NAME}\` → ${process.env.PREVIEW_URL} `
+ `<sub>(${sha})</sub>${dbToken}${flagsToken}`;
// Replace this app's block in the body (or append it), preserving any
// other app's block another matrix job may have already written.
// Author-constrained like every other lookup on this seam (#3740).
const isTrustedCi = (c) => c.user?.login === "github-actions[bot]" && c.user?.type === "Bot";
const upsert = (body) => {
const header = `${marker}\n### 🚀 Preview deployed`;
let b = body && body.includes(marker) ? body : header;
const re = new RegExp(
`${appMark.replace(/[.*+?^${}()|[\\]\\\\]/g, "\\\\$&")}\\n[^\\n]*`,
);
return re.test(b) ? b.replace(re, appLine) : `${b}\n${appLine}`;
};
// Retry the read-modify-write: a sibling matrix job may create or
// update the same comment between our read and write, so re-list and
// re-resolve `existing` each attempt — a leg that lost the create
// race observes the sibling's comment on its next pass and falls into
// the update branch instead of creating a duplicate (issue #273).
for (let attempt = 0; attempt < 5; attempt++) {
try {
const {data: comments} = await github.rest.issues.listComments({
...context.repo, issue_number: context.issue.number, per_page: 100,
});
const existing = comments.find((c) => isTrustedCi(c) && c.body?.includes(marker));
if (existing) {
await github.rest.issues.updateComment({
...context.repo, comment_id: existing.id, body: upsert(existing.body),
});
} else {
await github.rest.issues.createComment({
...context.repo, issue_number: context.issue.number, body: upsert(""),
});
}
return;
} catch (err) {
if (attempt === 4) throw err;
await new Promise((r) => setTimeout(r, 500 * (attempt + 1)));
}
}