Repository navigation
140 lines (133 loc) · 8.1 KB
/
Copy pathdeploy-stage.yml
File metadata and controls
140 lines (133 loc) · 8.1 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
# Deploy stage stack on the VPS over SSH.
#
# Required repo secrets:
# STAGE_HOST — VPS hostname or IP
# STAGE_USER — SSH user (e.g. root or deploy)
# STAGE_SSH_PASSWORD — SSH password for that user (store in GitHub Secrets; prefer SSH keys when possible)
# STAGE_DEPLOY_PATH — absolute path to clutch-deploy on the server (e.g. /root/clutch-deploy)
#
# Optional:
# STAGE_SSH_PORT — SSH port if not 22
#
# The server must: have Docker + Compose v2, git clone of this repo at STAGE_DEPLOY_PATH (for git pull),
# and for private GHCR images: docker login ghcr.io (run once on the VPS).
name: Deploy stage (VPS)
on:
workflow_dispatch:
inputs:
reset_chain:
description: "DESTRUCTIVE and rarely needed — leave UNTICKED for normal deploys. Runs `down -v`, wiping the chain, explorer DB, Grafana/Seq, the treasury and orchestrator databases, AND Bitcart's database. That last one invalidates BITCART_TOKEN/BITCART_STORE_ID in .env, so you must re-run 'Provision Bitcart (stage)' afterwards or the orchestrator will point at a store that no longer exists. Tick ONLY after changing consensus values in config/node/*.toml, since a new ChainInit genesis cannot import onto old chain data."
type: boolean
default: false
set_images:
description: "Optional. Move stage images to new tags before deploying: image=tag pairs separated by spaces, e.g. clutch-treasury=sha-239f538 clutch-orchestrator=sha-239f538 clutch-tron-signer=sha-239f538. The pin is committed to main first. Leave empty to deploy exactly what is pinned."
type: string
default: ""
push:
branches: [main]
paths:
- "docker-compose.yml"
- "docker-compose.stage.cloudflare-flex.yml"
- "docker-compose.treasury.yml"
- "config/**"
- ".github/workflows/deploy-stage.yml"
# The deploy logic itself lives here now. Without this, editing scripts/deploy-stage.sh
# changes what a deploy DOES but never triggers one — the fix sits in main looking merged.
- "scripts/**"
repository_dispatch:
types: [deploy-stage]
# No workflow-level concurrency: only the deploy job is serialized, below. GitHub keeps at most ONE
# pending run per concurrency group and cancels the older pending one, so with the whole workflow
# in the group, two image builds finishing close together would lose the first one's pin.
jobs:
# Every image is pinned to an exact tag, and a deploy ships exactly those tags
# (scripts/set-image.sh). An image workflow's dispatch carries its new tag as
# client_payload.set_images, and a manual run can pass the same string. This job commits that pin
# to main BEFORE anything deploys, so main always says what stage runs or is about to run.
# Nothing else moves: a deploy for one repo's image no longer ships what another repo built since.
pin:
if: ${{ (inputs.set_images || github.event.client_payload.set_images || '') != '' }}
runs-on: ubuntu-latest
permissions:
contents: write
steps:
- uses: actions/checkout@v4
with:
ref: main
- name: Commit the new tags to main
env:
SET_IMAGES: ${{ inputs.set_images || github.event.client_payload.set_images }}
run: |
set -euo pipefail
git config user.name "github-actions[bot]"
git config user.email "41898282+github-actions[bot]@users.noreply.github.com"
# Split on spaces, with globbing off. set-image.sh checks every word (a known image, a
# sha-<7> tag that ghcr.io has) before it writes anything, so nothing from the payload
# reaches sed or git unchecked. The push is made with GITHUB_TOKEN, which starts no
# workflow, so it cannot trigger a second deploy through the push paths above.
set -f
# shellcheck disable=SC2086
PUSH=1 bash scripts/set-image.sh stage $SET_IMAGES
deploy:
needs: pin
# After a successful pin, or when there was nothing to pin. Never after a failed one: that would
# ship the OLD tags from a run that asked for new ones, and still report a deploy.
if: ${{ !cancelled() && (needs.pin.result == 'success' || needs.pin.result == 'skipped') }}
runs-on: ubuntu-latest
# A newer pending deploy replacing an older one loses nothing: each pin is already on main, and
# the host pulls main before it deploys.
concurrency:
group: deploy-stage
cancel-in-progress: false
steps:
- name: Deploy via SSH
env:
SSHPASS: ${{ secrets.STAGE_SSH_PASSWORD }}
CLUTCH_SSH_HOST: ${{ secrets.STAGE_HOST }}
CLUTCH_SSH_USER: ${{ secrets.STAGE_USER }}
run: |
# Plain ssh, not appleboy/ssh-action. That action downloads a drone-ssh binary from
# GitHub releases on EVERY run -- unconditionally, there is no skip-if-present -- and on
# 2026-09-14 that CDN answered 504 to all six attempts of the action's own `curl --retry 5`
# and failed a deploy. Its retry is the thing we watched lose, so retrying harder is not
# the fix; not depending on a runtime download is. `ssh` is already on the runner.
#
# The script travels as a FILE, never as an argument. The 11 KB inline block this repo
# used to carry failed three times in ways that contradicted its own source, because what
# bash received was not what the workflow contained. One quoting layer is better than two.
set -euo pipefail
command -v sshpass >/dev/null || { sudo apt-get update -qq && sudo apt-get install -y -qq sshpass; }
R="/tmp/clutch-ci-${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}"
SSH_OPTS="-o StrictHostKeyChecking=no -o UserKnownHostsFile=/dev/null -o ConnectTimeout=30 -o ServerAliveInterval=30"
cat > "$RUNNER_TEMP/clutch-remote.sh" <<'CLUTCH_REMOTE_EOF'
set -euo pipefail
cd "${{ secrets.STAGE_DEPLOY_PATH }}"
# `origin main`, not a bare pull. Without a refspec this host resolved FETCH_HEAD to
# every branch it had just fetched and died with "Cannot fast-forward to multiple
# branches" — so pushing two feature branches was enough to stop the host updating.
#
# And no `|| echo ... continuing`. It swallowed exactly that failure: the deploy went on
# to run whatever scripts and compose files the host already had, reported success, and
# the one line saying the pull had failed was three hundred lines up the log. A deploy
# that silently ships the previous commit is worse than a deploy that stops.
git pull --ff-only origin main
# Everything else lives in scripts/deploy-stage.sh. Keep this wrapper tiny.
#
# The deploy logic used to be an 11 KB inline block here. It failed three times in a
# row in ways that contradicted the source — exit 1 with no message and no ERR trap,
# twice, then a syntax error at "line 370" of a 208-line script that passes bash -n.
# What bash received was not what the workflow contained; the block had outgrown the
# trip through YAML, the ssh-action and `bash -c`, and the failure point moved every
# time the text grew. Two wrong fixes came out of reading it as a logic bug.
#
# Run with `bash`, not the exec bit: `chmod +x` on a tracked file reads as a local
# modification and silently blocks `git pull --ff-only` on this host, which once kept
# four fixes off the server for an hour.
#
# git pull stays here, above the call, so the script that runs is the one just pulled.
RESET_CHAIN="${{ inputs.reset_chain }}" bash scripts/deploy-stage.sh
CLUTCH_REMOTE_EOF
sshpass -e ssh $SSH_OPTS "$CLUTCH_SSH_USER@$CLUTCH_SSH_HOST" "cat > $R.sh" < "$RUNNER_TEMP/clutch-remote.sh"
# Old files from earlier runs are swept here rather than in a cleanup step that
# a cancelled job would skip.
timeout 10m sshpass -e ssh $SSH_OPTS "$CLUTCH_SSH_USER@$CLUTCH_SSH_HOST" "find /tmp -maxdepth 1 -name 'clutch-ci-*' -mtime +1 -delete 2>/dev/null; bash $R.sh"