diff --git a/CHANGELOG.md b/CHANGELOG.md index 8fc0fb4..fa41a33 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -11,6 +11,33 @@ Notable changes to the Evomedia.net Token Savers. ## Unreleased ### Changed +- **`zdeploy` verification picks its channel on every retry, and never asks + the bare IP** — the check used to choose its channel once, before the wait + loop, by probing; the probes raced the app restart the check exists to wait + through, so the whole window went to the edge fallback. For a project with + no `domain` that fallback had no `Host` header, and the proxy can then only + answer from its **default vhost — a different product**: that is how one + deploy's check compared another app's build number against its own. Now + channels are re-resolved each retry in trust order (docker-network + `viaProxy` → `localhost:` → edge with the project's `Host`), the edge + is skipped entirely when there is no host to route by, and a project with + no trustworthy channel is reported as unverifiable instead of guessed at. + The final warning also says which failure happened: a version that never + matched (stale/failed build) reads differently from channels that never + answered (probably still booting). `verify.timeoutSeconds` joins the + config so a project that is slow to boot — e.g. one that runs database + migrations in its entrypoint — can widen its own window instead of + warning on every routine success. +- **An interrupted deploy can no longer destroy server-side `.env` files** — + the preserve/restore of operator files is transactional: the restore comes + from a tarball taken before the tree is replaced, so a deploy that dies + mid-flight leaves the previous files in place instead of an empty + directory. +- **A successful deploy no longer reports failure** — `docker compose + restart` writes routine progress to stderr, which PowerShell 5.1 turns + into a terminating error under `$ErrorActionPreference = 'Stop'`; four ssh + calls bypassed the wrapper that flattens this. All remote steps now run + through it and are judged by exit code alone. - **`zversion bump` is once per *release*, not once per PR** — the usage text and the `bump` help line both said "one per PR, one per defect fix". The build number names something that shipped, so a release carrying five PRs @@ -21,6 +48,14 @@ Notable changes to the Evomedia.net Token Savers. written. ### Added +- **`tests/VerifyPlan.Tests.ps1`** — the verification channel-selection rules + are pure functions in `ZHelpers.ps1` (`Get-VerifyAttempts`, + `Get-VerifyTimeout`) and Pester pins them, including "a project with no + domain must never produce an edge attempt" and the PowerShell 5.1 + one-element-unroll trap. +- **`scripts/readme_txt.py` and `README.txt`** — a generated plain-text twin + of the README for terminals and pagers. `README.txt` is generated, never + edited by hand. - **Read a live build from inside the docker network, not through the public proxy** — `zdeploy`, `zec2` and `zec2online` now prefer `docker exec curl http:///api/build-version` when a diff --git a/CHANGELOG.txt b/CHANGELOG.txt index e9ed815..49b5c99 100644 --- a/CHANGELOG.txt +++ b/CHANGELOG.txt @@ -10,6 +10,33 @@ Notable changes to the Evomedia.net Token Savers. Unreleased ---------- Changed +- zdeploy verification picks its channel on every retry, and never asks + the bare IP - the check used to choose its channel once, before the wait + loop, by probing; the probes raced the app restart the check exists to + wait through, so the whole window went to the edge fallback. For a + project with no domain that fallback had no Host header, and the proxy + can then only answer from its default vhost - a different product: that + is how one deploy's check compared another app's build number against + its own. Now channels are re-resolved each retry in trust order + (docker-network viaProxy -> localhost: -> edge with the project's + Host), the edge is skipped entirely when there is no host to route by, + and a project with no trustworthy channel is reported as unverifiable + instead of guessed at. The final warning also says which failure + happened: a version that never matched (stale/failed build) reads + differently from channels that never answered (probably still booting). + verify.timeoutSeconds joins the config so a project that is slow to + boot - e.g. one that runs database migrations in its entrypoint - can + widen its own window instead of warning on every routine success. +- An interrupted deploy can no longer destroy server-side .env files - the + preserve/restore of operator files is transactional: the restore comes + from a tarball taken before the tree is replaced, so a deploy that dies + mid-flight leaves the previous files in place instead of an empty + directory. +- A successful deploy no longer reports failure - docker compose restart + writes routine progress to stderr, which PowerShell 5.1 turns into a + terminating error under ErrorActionPreference = Stop; four ssh calls + bypassed the wrapper that flattens this. All remote steps now run + through it and are judged by exit code alone. - zversion bump is once per RELEASE, not once per PR - the usage text and the bump help line both said "one per PR, one per defect fix". The build number names something that shipped, so a release carrying five PRs moves @@ -20,6 +47,14 @@ Changed written. Added +- tests/VerifyPlan.Tests.ps1 - the verification channel-selection rules + are pure functions in ZHelpers.ps1 (Get-VerifyAttempts, + Get-VerifyTimeout) and Pester pins them, including "a project with no + domain must never produce an edge attempt" and the PowerShell 5.1 + one-element-unroll trap. +- scripts/readme_txt.py and README.txt - a generated plain-text twin of + the README for terminals and pagers. README.txt is generated, never + edited by hand. - Read a live build from inside the docker network, not through the public proxy — zdeploy, zec2 and zec2online now prefer docker exec curl http:///api/build-version when a diff --git a/CHECKSUMS.txt b/CHECKSUMS.txt index cbe2eed..ad33c3f 100644 --- a/CHECKSUMS.txt +++ b/CHECKSUMS.txt @@ -8,14 +8,14 @@ e35d91a175c29dcbe285b55397371a584000aab9ae3de6839f8b31def1d9dfd0 token-count.ps 49d48d55bab1b9ee5bc66cb96340a20fa50241b6a3558d4c518db3e06d4553e6 zchecksums.cmd e0cbc6f263c129d89e87e8e7356fb256f59f21a6a4b209e191e7c70a3611286e zchecksums.ps1 bd5563bf74b83423aca49a56450f3712a9aec3d6a59f8b7862d5c5988c6defd2 zdeploy.cmd -4a55ed714358b910fd2f326a6222e2ebaa99a1321df025a19edb6bbe0da13e76 zdeploy.ps1 +869efc42e86c4dd0c208e88598abcf9e3226afd5a4eb3fc4346ad81e933ef177 zdeploy.ps1 1883d7307682c68bf3786203f54e9abfe12fcb09e3856190e3a7af5fffb2a16a zec2.cmd f9d0406f97e19303d8b368df8aaf70e5011a99703f67ca7b2b526e82a3c68789 zec2.ps1 e1da88e4af6d5c88cc95231dc544ef29440d306e70739d0ba1f72795d359778d zec2_rotatekeys.cmd ea2fcf081491dbb98794bc2ef6ce41563fef8a7c3553840e75aff757ea59d582 zec2_rotatekeys.ps1 b30b10ca6d7930021f565a3d90eda967f1ca18366b30a671eab0668f6d5695eb zec2online.cmd 5b8c3bfc707255f5782ea6906075817f5bfe8d77fb1a4299a33581d0c70b9569 zec2online.ps1 -d550f9866c1f746da065021e5150566b3aebc07a51aa01ba6459f29560765b23 ZHelpers.ps1 +860e51adf035232b9242cff8866e31bb2dfe5117e393e0c698e958ef2f67d701 ZHelpers.ps1 1167972175b9d60613d1b8da55053c13a390037fd075fd8e411374d7a402ff72 zkill.cmd 0a4c09c2a85bfdd8577ee43179128fc6b14335e5232a2b172f4558542b8706f6 zkill.ps1 b81c06fc85045b69b298450b84757c1997ea14354e9774d579c27e41e2f43515 ZKiller.ps1 diff --git a/README.md b/README.md index ed4c0ee..8fdb50e 100644 --- a/README.md +++ b/README.md @@ -378,6 +378,21 @@ Give the project a `verify` block instead, and `zdeploy` checks the app **from t `port` is the host port the app publishes on the server; `path` defaults to `/`; `expect` is an optional substring the response must contain (an app version string makes this equivalent to build-number verification). Python-kind projects use `verify` automatically when there's no `build_version_tool.py` — and projects with *neither* a `domain` nor a `verify` block are now honestly reported as **NOT verified** instead of green-lighting the proxy's default page. +Two more keys make the check unambiguous and patient: + +```json +"verify": { + "port": 8005, "path": "/health", + "viaProxy": "my_edge_proxy", "upstream": "myapp_container:8000", + "timeoutSeconds": 120 +} +``` + +- **`viaProxy` + `upstream`** — read the version **over the docker network**, by exec-ing a curl inside the named container (usually the edge proxy, since it is on every stack's network). This is the most trustworthy channel there is: it cannot answer from the wrong product, and it works for apps that publish no host port at all. +- **`timeoutSeconds`** — how long verification may wait. An app that runs database migrations in its entrypoint exceeds a 30-second window *on every deploy that ships one*, and a warning that fires on routine success teaches you to ignore the one that matters. + +Verification walks its channels in trust order — docker-network `viaProxy`, then `localhost:`, then the edge with the project's own `Host` header — and **re-picks the channel on every retry**. That last part is the point: the check runs while the app is restarting, which is exactly when the good channels are briefly down, and locking the choice in up front is how a whole verification window gets spent asking the wrong thing. A project with no domain is **never** asked for via the edge: without a `Host` header the proxy can only answer from its default vhost — a different product — and no answer beats somebody else's answer. + `zec2` and `zec2online` use these same endpoints to show what's live and flag local/server version drift. --- diff --git a/README.txt b/README.txt new file mode 100644 index 0000000..23a3a96 --- /dev/null +++ b/README.txt @@ -0,0 +1,436 @@ + + +zscripts Token Savers +===================== + +When an AI coding agent orchestrates your infrastructure — starting dev servers, deploying to EC2, diagnosing 502s — it spends hundreds to thousands of tokens per operation on SSH plumbing, Docker output, and retry logic. Those tokens should go to code. + +Token Savers gives you short, one-word commands to run those parts yourself: zdeploy myapp, zrepair myapp, zstart myapp. You handle the deterministic infrastructure; your agent handles code. Running these scripts manually instead of asking your agent to orchestrate them keeps measured script output out of your agent's context window — ~26,500 tokens per active development day at a typical run cadence. Per-run figures are measured; the daily total applies typical run counts. See TOKEN_SAVINGS.md (TOKEN_SAVINGS.md) for the numbers and method. + +Every command is a tiny PowerShell script driven by a single JSON config file. The project key you define in that config is the command argument — add myapp to the config and zstart myapp, zdeploy myapp, zbackup myapp all just work, no script edits needed. + +Requirements: Windows, PowerShell 5.1+, OpenSSH client (ssh/scp, ships with Windows 10/11), and Docker + docker compose on the remote host for the deploy scripts. + +--- + +Install +------- + +No installer. Clone the repo and add the folder to your PATH: + + git clone https://github.com/evomedia-net/evo.zscripts.git C:\tools\zscripts + + # Add to your user PATH (new terminals pick it up automatically) + [Environment]::SetEnvironmentVariable( + "Path", + [Environment]::GetEnvironmentVariable("Path", "User") + ";C:\tools\zscripts", + "User" + ) + +Open a new terminal and every command below works from any directory. The .cmd wrappers invoke PowerShell with -ExecutionPolicy Bypass, so no execution-policy changes are needed — zstart myapp just works from cmd, PowerShell, or a VS Code terminal. + +--- + +Configure +--------- + +All machine-specific values (server IP, SSH user and key, folder paths, project definitions) live in one file: zconfig.json. It is gitignored — your secrets never leave your machine. + + cd C:\tools\zscripts + copy zconfig.example.json zconfig.json + notepad zconfig.json + +The example config ships with sample projects named by their kind — pyapp, viteapp, nextapp, edge, analytics. Rename the keys to your own project names; the key is what you type as the command argument. Add as many projects as you like — no script edits ever needed. + +Set the ZCONFIG environment variable to point at a config somewhere else — handy for a second machine profile, or for running against a scratch config without touching your real one. + +Config reference +---------------- + + { + "ec2": { + "ip": "203.0.113.10", // your server's public IP + "user": "youruser", // SSH user on the server + "pemKey": "C:\\Users\\You\\.ssh\\key.pem", // path to your SSH private key + "stackRoot": "/home/youruser/stack" // parent dir for all deployed projects + }, + "paths": { + "temp": "C:\\dev\\temp", // deploy zips staged here (auto-deleted) + "backupsLocal": "C:\\dev\\backups\\projects", // zbackup output + "backupsEc2": "C:\\dev\\backups\\ec2", // zbackup_ec2 output + "scriptsRoot": "C:\\tools\\zscripts", // this folder + "oneDriveBackups": "" // zsync destination ("" disables) + }, + "projects": { + "myapp": { + "label": "My App", // display name in output + "kind": "python", // python | vite | nextjs | edge | docker + "localRoot": "C:\\dev\\myapp", // project folder on this machine + "startModule": "myapp.main", // python kind: runs "python -m myapp.main" + // "startApp": "app.main:app", // ...or, for ASGI/FastAPI: uvicorn app.main:app --port --reload + "install": "-e .", // optional: pip args `zsetup` uses (auto-detects "-e ." / "-r requirements.txt") + "ports": { "dev": 8080, "prod": 3000 }, // local dev port / direct server port + "domain": "www.myapp.com", // public domain (health checks + verification) + "start": { // optional zstart pre-steps + "gitPull": true, // git pull --ff-only before starting + "env": { "MYAPP_DEBUG": "1" } // env vars for the dev server process + }, + "db": { "user": "dbuser", "name": "dbname" }, // optional: enables db dump/wait steps + "migrations": "prisma", // optional: run prisma migrate on deploy + "remote": { + "path": "/home/youruser/stack/myapp", // deploy target on the server + "composeDir": "/home/youruser/stack/myapp/docker", // optional: if compose isn't at path root + "appService": "app", // optional: compose service name override + "containerName": "myapp" // optional (vite): container to read build-version from + }, + "deploy": { + "zipName": "MyAppDeploy.zip", // optional: defaults to Deploy.zip + "gitPull": true, // optional: git pull --ff-only before zipping + "exclude": ["docs", "big-data-folder"] // optional: extra top-level dirs/files to skip + } + } + } + } + +Optional blocks do real work: + +- install — how zsetup installs a python project's dependencies into its .venv: the pip args, e.g. "-e .", "-e backend" (deps in a subfolder), or "-r requirements.txt". Omit it and zsetup auto-detects a root pyproject.toml/setup.py (-e .) or requirements.txt (-r requirements.txt). zstart never installs — run zsetup once, then zstart . +- start — pre-start steps for zstart: gitPull: true runs git pull --ff-only in the project root first (never starts a stale checkout), and env sets environment variables for the dev-server process (feature flags, reload switches). +- db — deploys wait for pg_isready and zbackup_ec2 pulls a pg_dump, both against the compose service named db. Omit it and those steps are skipped cleanly. +- deploy.gitPull — git pull --ff-only in the project root before zipping, so a merged PR actually ships. Since zdeploy zips your working tree, a checkout left behind origin would otherwise deploy stale code and still bump the build number — a silent no-op that looks like success. A failed pull (dirty tree that conflicts, diverged history) aborts the deploy rather than shipping uncertain code. +- migrations": "prisma" — runs npx prisma migrate deploy inside the app container after each deploy. +- Compose service-name conventions — handlers assume the app service is named app (python) or web (nextjs) and the database service db. Override the app service with remote.appService. +- Edge extras — an edge-kind project can set proxyContainer (the nginx container's name, used for reloads and stale-container cleanup) and certsSource (a host path with TLS certs, mounted read-only when validating nginx.conf). + +--- + +Commands +-------- + +The .cmd wrappers are the everyday interface. Every command takes one or more project keys from your config; several also accept all. A leading dash is tolerated (zdeploy -myapp works the same as zdeploy myapp) for anyone with switch-style muscle memory. + +| Command | What it does | +|---|---| +| zsetup ... | Create the project's Python venv + install deps (or npm install for node) | +| zstart ... | Start local dev server(s) | +| zstartd ... | Same, detached (new window, returns immediately) | +| zkill ... \| all | Kill local dev server(s) by port | +| zrestart ... | Kill + start in one step | +| zrestartd ... | Kill + start detached | +| zdeploy ... \| all | Zip → upload → rebuild → verify a project on the server | +| zec2 [ ...] | Quick reachability check (TCP + HTTP + live build version) | +| zec2online [ ...] | Deep health check; auto-starts downed stacks, streams diagnostics | +| zrepair ... | Audit + repair compose/proxy state on the server | +| zec2_rotatekeys | Rotate/reset secret keys in a project's server-side .env (values generated server-side; never printed) | +| zbackup ... \| all | Zip local project sources (+ DB dump) to the backups folder | +| zbackup_ec2 [ ...] | Pull DB dumps + server-side data files down from the server | +| zsync [] | Copy new backups offsite (or build + mirror a vite dist) | +| zstart_docker | Run a local docker compose stack from scriptsRoot\docker\ | +| zchecksums [-Update] | Verify every script against CHECKSUMS.txt (SHA-256) | +| zversion [bump \| bump-stage \| set ] | Show or advance the toolkit version (stamps every header) | +| zrelease [-Verify] | Package the current version as releases/zscripts-.zip + .sha256 | + +Local development +----------------- + +zstart — start dev servers +-------------------------- + + zstart [ ...] [-Port N] [-BindHost ] [-Detached] + +Starts each project's dev server using the handler for its kind: python runs python -m — or, for an ASGI/FastAPI app, uvicorn (e.g. app.main:app) with the dev port and --reload — preferring the project's .venv; vite runs npm run dev -- --host --port, nextjs runs npm run dev with PORT set. Runs npm install automatically if node_modules is missing. A project's optional start config block runs first — gitPull fast-forwards the checkout and env sets process environment variables. Two more opt-in conveniences: if the project has a motd/ folder of .txt files, one is shown (rotating) at startup; if it has scripts/build_version_tool.py, the build number is bumped on each start. + + zstart viteapp # dev server on its configured port + zstart pyapp -Port 9000 # override the port + zstart viteapp -BindHost 0.0.0.0 # expose on the LAN + zstartd nextapp # detached: window opens, prompt returns + +zkill — stop dev servers +------------------------ + + zkill [ ...] | all [-Port N] [-KillAll] + +Finds whatever is LISTENING on each project's dev port and kills it — along with its whole process tree, children first. That matters for auto-reloading servers (uvicorn/watchfiles, nodemon): their worker processes inherit the listening socket and would otherwise survive as orphans, serving stale code. -KillAll also hunts down stray node/python/next-server processes whose command line references the project folder. all targets every project that has a dev port — the one-shot "stop everything I've got running locally". + + zkill viteapp # free the port + zkill pyapp viteapp nextapp # nuke everything + zkill all # stop every project's dev server + zkill nextapp -KillAll # also kill orphaned runtime processes + +zrestart — kill then start +-------------------------- + + zrestart [ ...] [-Port N] [-KillAll] [-NoRestart] [-Detached] [-BindHost ] + +The "it's wedged, bounce it" command: kill phase, then start phase with the same flags. -NoRestart makes it kill-only; zrestartd restarts detached. + +Server deployment & operations +------------------------------ + +zdeploy — deploy to the server +------------------------------ + + zdeploy [ ...] [-Note "message"] + zdeploy all [-Note "message"] + +The core workflow, per project kind (projects with deploy.gitPull first git pull --ff-only so a merged PR isn't left behind): + +- python / vite / nextjs — zip the local source (excluding .git, node_modules, envs, archives, junk, plus anything in deploy.exclude), free disk space on the server (docker prune; aborts if under 1.5 GB free), scp the zip up, unzip into remote.path preserving all server-side .env* files plus anything listed in deploy.preserve (staged keys, certs, seed data — files or directories), docker compose build + up -d, then verify the live site reports the new build version (see Enabling deploy verification (#enabling-deploy-verification)). nextjs additionally waits for Postgres (db block) and applies migrations (migrations field). Zips are always deleted locally afterward. +- edge — uploads every top-level file in the edge folder (nginx.conf, compose, css, htpasswd, …), validates the new config with nginx -t before switching over, then recreates the proxy. +- docker — uploads the compose folder's files, docker compose pull + up -d. For stacks that run stock images (analytics, mail, etc.). + +all deploys every project — edge kinds first, then the rest in config order — and stops at the first failure. + + zdeploy viteapp + zdeploy pyapp -Note "fix billing banner" + zdeploy all -Note "weekly release" + +zec2 — reachability check +------------------------- + + zec2 [ ...] # no args = every project with a domain + +For each project: TCP connect, then an HTTP GET with the project's domain as the Host header, then the live build version. Fast "is it up?" answer with firewall hints when it isn't. + +zec2online — health check with auto-recovery +-------------------------------------------- + + zec2online [ ...] # no args = every project with a domain + +The heavier sibling: verifies each app over HTTP, compares the local build version against what the server is actually serving (a mismatch means "redeploy?" — or a stale cache), and if a site is down it SSHes in, runs docker compose up -d for the app and the edge proxy, waits up to 30 s, and streams compose logs and system diagnostics if recovery fails. + +zrepair — fix server routing +---------------------------- + + zrepair [ ...] + +Validates the edge proxy's nginx config (if an edge project is defined), shows each stack's compose status, starts anything that's down, and smoke-tests the live domain. For the "deploy succeeded but the site 502s" class of problem. + +zstop.ps1 — stop server stacks +------------------------------ + + zstop [ ...] + +docker compose down for the selected stacks on the server. Data volumes are preserved; zdeploy brings a stack back. (PowerShell script only, no .cmd wrapper.) + +zec2_rotatekeys — rotate server-side secrets +-------------------------------------------- + + zec2_rotatekeys [-Rotate KEY,KEY] [-Set KEY,KEY] [-EnvFile rel/path] [-Restart] [-WhatIf] + +For when a secret leaks or a deploy overwrites a production .env with dev values: rotate or reset keys in a project's server-side .env without the values ever passing through this machine's shell history, a command argument, or your screen. -Rotate keys are regenerated on the server with openssl rand -hex 32 — the new value is written straight into the .env there and never leaves the box. -Set keys are typed into a masked prompt and streamed to the server over SSH stdin (never a command argument, never echoed), for operator-known values like DATABASE_URL or ADMIN_EMAIL. The current server .env is copied to a timestamped .bak before any change; the KEY line is updated atomically, matching an existing key or appending it. The env file is auto-detected from the project's deploy.preserve (first *.env) or defaults to .env — override with -EnvFile backend/.env. Nothing touches the running app unless you pass -Restart, which recreates the container (docker compose up -d --force-recreate ) so it actually reloads the new .env — a plain restart would keep the old environment. Being high-impact, it confirms before writing; -WhatIf prints the exact plan and changes nothing. + + # Preview only — see exactly what would change, change nothing: + zec2_rotatekeys pyapp -Rotate JWT_SECRET -Set DATABASE_URL,ADMIN_EMAIL -WhatIf + + # Regenerate the JWT secret, restore the operator-known values, then restart: + zec2_rotatekeys pyapp -Rotate JWT_SECRET -Set DATABASE_URL,ADMIN_EMAIL,ADMIN_PASSWORD -Restart + +Backups +------- + +zbackup — local backups +----------------------- + + zbackup [ ...] [-Tag "label"] + zbackup all # every project + this scripts folder + zbackup scripts # just this scripts folder ('scripts' is reserved) + +Zips each project's source into paths.backupsLocal\\_[_tag].zip. If the project's .env declares a DATABASE_URL, a Postgres dump is bundled into the zip automatically — quoted values (Prisma-style), postgres:///postgresql+driver:// schemes, and URLs without an explicit port all parse. backend\.env is checked too, for frontend/backend split projects. -Tag labels the archive — handy before risky changes. + + zbackup all # everything + zbackup pyapp -Tag "pre-migration" + +zbackup_ec2 — pull backups from the server +------------------------------------------ + + zbackup_ec2 [ ...] # no args = every project with a remote.path + +For projects with a db block, runs pg_dump inside the server's db container. Also zips server-side data dirs (uploads/, archive/, dist/) when present, then downloads everything to paths.backupsEc2 and cleans up the remote temp files. + +zsync — sync backups offsite +---------------------------- + + zsync # new backup files -> paths.oneDriveBackups + zsync -Destination # npm run build, then mirror dist/ to path + +The no-args mode copies only files that don't already exist at the destination (never overwrites, never deletes). The project mode is for mirroring a static build; it also honors $env:ZSYNC_DEST. + +zbackup_and_sync.ps1 — both in one +---------------------------------- + + zbackup_and_sync.ps1 [ ...] | all + +Runs zbackup, then zsync. This is what the scheduled task calls (with all). + +setup_backup_schedule.ps1 — nightly automation +---------------------------------------------- + +Run as Administrator once. Creates a Windows Scheduled Task that runs zbackup_and_sync.ps1 all daily at 2:00 AM. + +Utilities +--------- + +zstart_docker — local compose stack +----------------------------------- + + zstart_docker [-Build] [-Attached] [-Solo] + +Brings up a docker compose stack from scriptsRoot\docker\docker-compose.yml (or docker-compose.solo.yml with -Solo). Checks that Docker Desktop is actually running and tells you how to unwedge it if not. + +zsetup_mail.ps1 — provision mail accounts +----------------------------------------- + + zsetup_mail.ps1 -domain yourdomain.com [-mailHost mail.yourdomain.com] + +Creates admin@ and noreply@ mailboxes (with generated passwords) in a docker-mailserver container on the server, then prints the exact DNS records (MX, SPF, A) and SMTP/IMAP settings to plug into your app. + +ZHelpers.ps1 — shared library +----------------------------- + +Not run directly. Dot-sourced by the other scripts; provides config loading (Get-ZConfig, Get-ZProject), SSH helpers (Invoke-Ec2Step), the deploy/backup archiver (New-ProjectArchive), and process-kill helpers. Extend here if you're adding your own scripts. + +--- + +Enabling deploy verification +---------------------------- + +Why not just check for HTTP 200? Because a 200 proves nothing — a stale cached build serves 200 all day. These scripts verify a deploy by comparing build numbers: your app exposes its build version, the deploy expects to see the new number live, and a mismatch means the upload or Docker build failed (or you're looking at a cached build). + +It's optional — deploys still work without it, ending in a WARNING instead of a PASS — but it's the difference between "the server answered" and "the code I just shipped is actually running." + +1. Add a version file to your project +------------------------------------- + + // build-version.json (vite: in public/ · nextjs: in public/ · python: project root) + { "productVersion": "1.0", "buildNumber": 42 } + +2. Bump it during the server-side Docker build +---------------------------------------------- + +The convention: each deploy's Docker build increments buildNumber by one, so the deploy script expects local buildNumber + 1 to show up live. One line in your Dockerfile does it: + + RUN node -e "const f='public/build-version.json',v=require('./'+f);v.buildNumber++;require('fs').writeFileSync(f,JSON.stringify(v))" + +3. Expose it +------------ + +Vite / static sites — nothing to do: public/build-version.json is served at /build-version.json, which is where verification looks. (If your edge proxy blocks it from outside, set remote.containerName in config and verification reads it inside the container instead.) + +Next.js — add an API route at /api/build-version: + + // app/api/build-version/route.ts + import { NextResponse } from "next/server"; + import bv from "@/public/build-version.json"; + + export async function GET() { + return NextResponse.json({ build_version: `v${bv.productVersion}.${bv.buildNumber}` }); + } + +Python (FastAPI shown; any framework works) — expose /api/build-version: + + import json, pathlib + + @app.get("/api/build-version") + def build_version(): + bv = json.loads(pathlib.Path("build-version.json").read_text()) + return {"build_version": f"v{bv['productVersion']}.{bv['buildNumber']}"} + +Python projects can go further with a scripts/build_version_tool.py supporting get / set / bump subcommands — if present, zdeploy bumps the version inside the running container, records it, and zstart bumps on every dev start. + +Alternative: server-side health check (verify block) +---------------------------------------------------- + +Not every stack is published through the edge proxy — internal APIs, apps whose host port the firewall blocks, services waiting on a DNS record. For those, the old fallback (GET http:///) was worse than nothing: the edge proxy's default vhost answers with a 200 and the deploy "passes" even if your app never started. + +Give the project a verify block instead, and zdeploy checks the app from the server itself over SSH: + + "verify": { "port": 8005, "path": "/health", "expect": "\"status\":\"ok\"" } + +port is the host port the app publishes on the server; path defaults to /; expect is an optional substring the response must contain (an app version string makes this equivalent to build-number verification). Python-kind projects use verify automatically when there's no build_version_tool.py — and projects with neither a domain nor a verify block are now honestly reported as NOT verified instead of green-lighting the proxy's default page. + +Two more keys make the check unambiguous and patient: + + "verify": { + "port": 8005, "path": "/health", + "viaProxy": "my_edge_proxy", "upstream": "myapp_container:8000", + "timeoutSeconds": 120 + } + +- viaProxy + upstream — read the version over the docker network, by exec-ing a curl inside the named container (usually the edge proxy, since it is on every stack's network). This is the most trustworthy channel there is: it cannot answer from the wrong product, and it works for apps that publish no host port at all. +- timeoutSeconds — how long verification may wait. An app that runs database migrations in its entrypoint exceeds a 30-second window on every deploy that ships one, and a warning that fires on routine success teaches you to ignore the one that matters. + +Verification walks its channels in trust order — docker-network viaProxy, then localhost:, then the edge with the project's own Host header — and re-picks the channel on every retry. That last part is the point: the check runs while the app is restarting, which is exactly when the good channels are briefly down, and locking the choice in up front is how a whole verification window gets spent asking the wrong thing. A project with no domain is never asked for via the edge: without a Host header the proxy can only answer from its default vhost — a different product — and no answer beats somebody else's answer. + +zec2 and zec2online use these same endpoints to show what's live and flag local/server version drift. + +--- + +Adding a new project +-------------------- + +1. Add a key under projects in zconfig.json — copy the sample of the matching kind and rename it. +2. That's it: zstart, zkill, zrestart, zbackup, zdeploy, zec2, zec2online, zrepair, zstop all accept the new key immediately. +3. A project whose deploy doesn't fit the python/vite/nextjs/edge/docker patterns needs its own Invoke-Deploy function in zdeploy.ps1 — copy an existing handler; they're all variations on zip → upload → compose up → verify. + +Downloading without cloning +--------------------------- + +Each release is packaged as a zip in releases/ (releases/) — grab the latest zscripts-v*.zip, check it, unzip, done: + + sha256sum -c zscripts-v1.0.0.0.0.zip.sha256 # verify the download + unzip zscripts-v1.0.0.0.0.zip -d zscripts # extract + cd zscripts && sha256sum -c CHECKSUMS.txt # verify the contents + +The zip contains every command, CHECKSUMS.txt, zconfig.example.json, and the docs. Versions follow v{major}.{rc}.{beta}.{alpha}.{build}; every script header carries the release version it shipped in, so even a single copied file can be traced to its release. + +Verifying what you downloaded +----------------------------- + +CHECKSUMS.txt holds a SHA-256 for every .ps1 and .cmd in the repo. Check them before running anything: + + zchecksums + +Or with the standard tool on Linux/macOS/WSL — the manifest is sha256sum format: + + sha256sum -c CHECKSUMS.txt + +The hashes are identical on every platform: .gitattributes pins .ps1/.cmd to CRLF everywhere, so a file is byte-for-byte the same whether you cloned on Windows or Linux. + +zchecksums flags three things — a file whose contents changed, a listed file that's gone, and a script on disk that isn't in the manifest (so something added quietly still gets noticed). It exits non-zero on any of them. + +If you edit a script yourself, regenerate and commit the manifest with it: + + zchecksums -Update + +What this does and doesn't prove. CHECKSUMS.txt lives in the same repo as the scripts, so anyone who could alter a script could alter the manifest too. It's an integrity check, not a signature: it reliably catches a truncated clone, a local edit you forgot about, or a file added outside a commit. It does not prove the code came from this project — for that you'd need a signature or a hash published outside this repo. + +Tests +----- + +The toolkit has its own Pester (https://pester.dev) suite covering the pure logic — the exclude lists, config lookups, and version-label formatting that the deploy and backup paths depend on: + + Invoke-Pester .\tests + +Needs Pester 5+ (Install-Module Pester -Scope CurrentUser); Windows ships 3.x, which won't run these. The suite injects a fixture config through ZCONFIG, so it never reads your real zconfig.json and runs fine on a machine that has never been configured. + +The high-value case is the deploy-vs-backup split: deploys must exclude .env files and uploads/, backups must keep them. Get that backwards in either direction and you either ship secrets to production or quietly write backups that can't restore — neither fails loudly at runtime. + +Troubleshooting +--------------- + +- "zconfig.json not found" — you haven't copied zconfig.example.json yet. Every script tells you this and exits. +- "Unknown project key" — the argument doesn't match a key in zconfig.json; the error lists the valid keys. +- "PEM key not found" — fix ec2.pemKey in zconfig.json. +- Deploy aborts with "less than 1.5 GB free" — the server's disk is full even after auto-pruning. Grow the volume, or SSH in and run sudo docker system prune -af. +- Deploy ends with a version WARNING — the new build isn't what's being served: check the Docker build output, and see Enabling deploy verification (#enabling-deploy-verification) if you haven't set it up. +- Port already in use when starting — zkill first, or just use zrestart. + +License +------- + +MIT (LICENSE) diff --git a/ZHelpers.ps1 b/ZHelpers.ps1 index cb58c66..6b68892 100644 --- a/ZHelpers.ps1 +++ b/ZHelpers.ps1 @@ -15,8 +15,8 @@ $script:ArchiveExtensions = @( # exempt from ArchiveExtensions. A project that vendors a dependency as # vendor/*.tgz (common when a bundler cannot resolve `file:` links outside the # project root) needs that tarball in the deploy zip - dropping it makes a -# Dockerfile's `COPY vendor ./vendor` fail at image build, which is a confusing -# way to discover the archive filter ate a required build input. +# Dockerfile's `COPY vendor ./vendor` fail at image build, which is a +# confusing way to discover the archive filter ate a required build input. $script:ArchiveKeepDirNames = @('vendor') $script:ScriptExtensions = @('.ps1', '.cmd', '.bat') $script:JunkExtensions = @( @@ -42,8 +42,9 @@ $script:JunkDirNames = @( $script:ZConfigCache = $null # Where zconfig.json lives. Defaults to next to the scripts; override with the -# ZCONFIG environment variable. Useful for pointing a run at an alternate -# config, and it is the seam the test suite uses to inject a fixture. +# ZCONFIG environment variable, matching the bash port (zhelpers.sh does the +# same). Useful for pointing a run at an alternate config, and it is the seam +# the test suite uses to inject a fixture. function Get-ZConfigPath { if ($env:ZCONFIG) { return $env:ZCONFIG } return (Join-Path $PSScriptRoot "zconfig.json") @@ -122,18 +123,19 @@ function Get-Ec2Home { return "/home/$((Get-ZConfig).ec2.user)" } -# Options every deploy-path ssh/scp carries. Splat with @sshOpts. +# Options every deploy-path ssh/scp carries. Splat with @sshOpts, matching the +# idiom in zsetup_mail.ps1 and zec2_rotatekeys.ps1. # # BatchMode=yes is the one that matters. Without it ssh PROMPTS - for a # passphrase, a password, a sudo password - and waits forever. The deploy pipes # stderr into the pipeline (2>&1 | ForEach-Object) so the prompt is swallowed # on its way to the screen: the run simply stops under whatever step label was # printed last, with nothing to explain it and no obvious reason why that -# particular step would be slow. One deploy appeared to hang on "ensure shared -# web network", a step whose entire body is `docker network create web +# particular step would be slow. `zdeploy umami` appeared to hang on "ensure +# shared web network", a step whose entire body is `docker network create web # 2>/dev/null || true` against a network that already existed. # -# There is no prompt here you would ever want to answer - a deploy key is +# There is no prompt here we would ever want to answer - the deploy key is # unencrypted and sudo on the box is passwordless - so failing immediately is # strictly better than waiting on input that is never coming. # @@ -147,16 +149,16 @@ function Get-Ec2Home { # reads its stdin and forwards it to the remote command, and under PowerShell it # inherits the console handle - so it can block forever waiting on input nobody # is going to type. The timeouts above cannot help: they bound a connection that -# is dying, and this one was never established. One deploy stopped under -# "ensure unzip installed", a step whose body short-circuits when unzip is -# already present; the server showed no ssh session at all (`who` empty, no -# docker build running), which is what a client-side stdin block looks like -# from the other end. The zip had uploaded and prod stayed a release behind. +# is dying, and this one was never established. A deploy on 2026-08-19 stopped +# under "ensure unzip installed", a step whose body short-circuits on an already +# installed unzip; the server showed no ssh session at all (`who` empty, no +# docker build running), which is what a client-side stdin block looks like from +# the other end. The zip had uploaded and prod stayed a PR behind. # # Safe here because nothing that pipes stdin INTO ssh uses these options: -# zdeploy passes only command strings. Any script that DOES pipe into ssh must -# build its own option array - adding -n to those would break them, so do not -# hoist this beyond the deploy path. +# zdeploy passes only command strings, and the scripts that do pipe +# (one-off registration and key-rotation scripts) build their own +# option arrays. Adding -n to those would break them - do not hoist it. function Get-Ec2SshOpts { return @( '-n', @@ -170,9 +172,9 @@ function Get-Ec2SshOpts { # The same options for scp, which does NOT accept -n: OpenSSH's scp exits 1 with # "unknown option -- n" and prints its usage block. That failure is easy to -# misread, because the caller's own error text is what the operator sees while -# the usage text scrolls past above it - one deploy reported "Likely server disk -# space" on a box with plenty of room. +# misread, because the caller's own error text is what the operator sees and the +# usage text scrolls past above it - a deploy on 2026-08-19 reported "Likely +# server disk space" while the box sat at 79% with 8.1 GB free. # # Derived from Get-Ec2SshOpts rather than duplicated, so the timeouts can never # drift apart between the two transports. @@ -217,11 +219,11 @@ function Invoke-Ec2Step { # # Deploys ship the default branch, so this SWITCHES to it rather than pulling # whatever branch happens to be checked out. The old behaviour pulled the -# current branch, which breaks as soon as the remote deletes branches on merge: -# a checkout still sitting on its just-merged PR branch pulls a ref the merge -# deleted, and the deploy dies on "no such ref was fetched". Worse, when the ref -# DID still exist, pulling the feature branch meant a deploy could ship a -# branch rather than the default. +# current branch, which broke the day delete_branch_on_merge went on +# fleet-wide (2026-08-07): a repo still sitting on its just-merged PR branch +# pulls a ref the merge deleted, and the deploy dies on "no such ref was +# fetched". Worse, when the ref DID still exist, pulling the feature branch +# meant a deploy could ship a branch, not main. # # The switch refuses to run over local changes: a dirty tree aborts the deploy # with the file list rather than risk tangling uncommitted work. The stale @@ -328,32 +330,36 @@ function Get-ServerSideVersionCommand { $null when the project has not configured one. .DESCRIPTION - Reading a live build number through the public proxy only works while - that endpoint IS public - and a build stamp is something many sites - deliberately do not serve to the world. Blocking it at the proxy - without moving the readers first leaves every tool quietly reporting - "unknown", which looks identical to "could not reach it". + Four tools read a live build number, and all four did it by asking + the public edge: zdeploy's post-deploy check, zec2, zec2online, and + bash/zhelpers.sh. That works only for as long as the endpoint is + public, and it should not be: www's /build-version.json has been + blocked at the edge since the 2026-05-29 security pass, and evo.ehs + answering /api/build-version to anyone is the inconsistency this + closes. - Going through the proxy is also how a check reads the WRONG service: - the proxy answers from whichever vhost matches the Host header, so a - container with no public route gets another site's version back. + Going through the edge is also how a check reads the WRONG product. + The proxy answers from whichever vhost matches the Host header, so a + service with no public route gets somebody else's version back -- + evo-ai's deploy check compared evo.ehs's build against its own and + reported a failure on a deploy that had worked (evo.scripts#101). - A service reached only on a shared docker network cannot be curled - from the host when it publishes no port. It IS reachable by name from - another container on that network, which also exercises the real HTTP - path - so this proves the app is serving, not merely that its + A project reached only on the shared docker network cannot be curled + from the host: evoehs_app publishes no port. It IS reachable by name + from another container on that network, which also exercises the real + HTTP path -- so this proves the app is serving, not merely that its database knows a version. Config, on the project's `verify` block: "verify": { "path": "/api/build-version", - "viaProxy": "edge_proxy_container", - "upstream": "app_container:80" + "viaProxy": "evo_edge_proxy", + "upstream": "evoehs_app:80" } Returns $null when either key is missing, so every project without - this config keeps the behaviour it has today. + this config keeps exactly the behaviour it has today. #> param($Proj) @@ -370,9 +376,10 @@ function Get-LabelFromVersionJson { The build label out of a version endpoint's JSON text, or $null. .DESCRIPTION - Apps disagree about the field name - some answer `build_version`, - others `version`. Both mean "the build that is live", so both are - accepted rather than making an app rename its own field. + Two field names in the fleet: evo.ehs answers `build_version` on + /api/build-version, evo-ai answers `version` on /health. Both mean + "the build that is live", so both are accepted rather than making an + app rename its own field. #> param([string]$Text) @@ -832,3 +839,86 @@ function Stop-ZTracking { Remove-Item -LiteralPath $tp -Force -ErrorAction SilentlyContinue Write-ZTrailer -FinalNote $FinalNote } + +# ── Deploy-verification planning (pure; unit-tested in tests/) ────────────── + +function Get-VerifyTimeout { + <# + .SYNOPSIS + Seconds the live-build verification may wait, per project. + + .DESCRIPTION + verify.timeoutSeconds in zconfig.json lets a project that is slow to + BOOT say so, instead of every deploy of it warning on a success. An + app that runs database migrations in its entrypoint exceeds a 30s + window on every deploy that ships one - and a warning that fires on + routine success trains people to ignore the one that matters + (evo.scripts#101). + #> + param($Proj, [int]$DefaultSec) + if ($Proj.verify -and $Proj.verify.timeoutSeconds) { + return [int]$Proj.verify.timeoutSeconds + } + return $DefaultSec +} + +function Get-VerifyAttempts { + <# + .SYNOPSIS + The ordered ways to read this project's live build, most-trustworthy + first. Pure: config in, plan out - so the ordering rules are testable + without ssh. + + .DESCRIPTION + Three channels exist, and their order is the whole point (#101): + + exec - docker-network read via verify.viaProxy/upstream. Cannot + answer from the wrong product, works for apps with no + published port. + port - localhost: on the server. Same-box, still + unambiguous; used only if the body carries a version. + edge - http:// with a Host header. The proxy answers from + whichever vhost MATCHES that header, so without one this + channel can only reach the default vhost - which is a + different product (that is how evo-ai's check once read + evo.ehs's build number). It is therefore included ONLY + when the project has a host to route by, and never + otherwise: no answer at all beats somebody else's answer. + + The caller must walk this list EVERY retry, not once up front: the + verification runs while the app is restarting, which is exactly when + the good channels are briefly down. Deciding the channel before the + wait loop is how the whole window got spent on the worst one. + #> + param($Proj, [string]$ExecCmd) + $attempts = @() + if ($ExecCmd) { + $attempts += [pscustomobject]@{ + Kind = 'exec' + Label = "docker network: $($Proj.verify.upstream) (via $($Proj.verify.viaProxy))" + } + } + if ($Proj.verify -and $Proj.verify.port) { + $vPath = "/api/build-version" + if ($Proj.verify.path) { $vPath = [string]$Proj.verify.path } + $attempts += [pscustomobject]@{ + Kind = 'port' + Port = [int]$Proj.verify.port + Path = $vPath + Label = "server localhost:$($Proj.verify.port)$vPath" + } + } + $verifyHost = $null + if ($Proj.deploy -and $Proj.deploy.verifyHost) { $verifyHost = [string]$Proj.deploy.verifyHost } + elseif ($Proj.domain) { $verifyHost = [string]$Proj.domain } + if ($verifyHost) { + $attempts += [pscustomobject]@{ + Kind = 'edge' + HostHeader = $verifyHost + Label = "edge with Host: $verifyHost" + } + } + # The comma stops PS 5.1 unrolling a one-element array into a bare + # object - the same pipeline trap that deadlocked ztests day 2. + return ,$attempts +} diff --git a/scripts/readme_txt.py b/scripts/readme_txt.py new file mode 100644 index 0000000..4c802a0 --- /dev/null +++ b/scripts/readme_txt.py @@ -0,0 +1,76 @@ +# Evomedia.net Token Savers — https://github.com/evomedia-net/evo.zscripts +# Created by Kelly Michels · dev@evomedia.net +# Licensed under the MIT License. See LICENSE. + +"""Render README.md to README.txt with the markdown markup removed. + +README.txt exists for terminals, pagers and anywhere markdown doesn't +render. It is generated - never edit it by hand: + + python scripts/readme_txt.py # rewrite README.txt + python scripts/readme_txt.py --check # exit 1 if it is out of sync + +The test suite runs --check, so a README.md edit that forgets to +regenerate fails CI rather than shipping a stale mirror. +""" + +from __future__ import annotations + +import re +import sys +from pathlib import Path + +ROOT = Path(__file__).resolve().parent.parent + + +def _inline(text: str) -> str: + text = re.sub(r"!\[([^\]]*)\]\([^)]*\)", r"\1", text) # images -> alt text + text = re.sub(r"\[([^\]]+)\]\(([^)]+)\)", r"\1 (\2)", text) # links -> text (url) + text = re.sub(r"\*\*([^*]+)\*\*", r"\1", text) # bold + text = re.sub(r"(? str: + out: list[str] = [] + in_fence = False + for line in md.splitlines(): + if line.lstrip().startswith("```"): + # Drop the fence markers; the code itself stays, indented so it + # still reads as a block without the backticks. + in_fence = not in_fence + continue + if in_fence: + out.append((" " + line) if line else "") + continue + heading = re.match(r"^(#{1,6})\s+(.*)$", line) + if heading: + text = _inline(heading.group(2)) + out.append(text) + out.append(("=" if len(heading.group(1)) == 1 else "-") * len(text)) + continue + out.append(_inline(line)) + text = "\n".join(out) + text = re.sub(r"\n{3,}", "\n\n", text) + return text.strip() + "\n" + + +def main() -> int: + source = (ROOT / "README.md").read_text(encoding="utf-8") + rendered = render(source) + target = ROOT / "README.txt" + if "--check" in sys.argv: + current = target.read_text(encoding="utf-8") if target.exists() else "" + if current != rendered: + print("README.txt is out of sync - run: python scripts/readme_txt.py") + return 1 + print("README.txt is in sync") + return 0 + target.write_text(rendered, encoding="utf-8", newline="\n") + print(f"Wrote {target} ({len(rendered.splitlines())} lines)") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/VerifyPlan.Tests.ps1 b/tests/VerifyPlan.Tests.ps1 new file mode 100644 index 0000000..3d8bce3 --- /dev/null +++ b/tests/VerifyPlan.Tests.ps1 @@ -0,0 +1,96 @@ +# Deploy-verification planning (#101). +# +# Invoke-Pester .\tests +# +# The wrong-vhost failure was never about parsing a response - it was about +# WHICH channel got asked. So the channel-selection rules live in a pure +# function (Get-VerifyAttempts) and are pinned here, where they can be tested +# without an EC2 box: a project with no domain must never produce an edge +# attempt, because an edge request with no Host header can only reach the +# default vhost - which is a different product. That exact gap read evo.ehs's +# build number during evo-ai deploys twice on 2026-08-31 alone. +# +# ZHelpers.ps1 is dot-sourced rather than zdeploy.ps1: zdeploy executes its +# main flow on load, helpers only define functions. + +BeforeAll { + . (Join-Path (Split-Path -Parent $PSScriptRoot) "ZHelpers.ps1") + + function New-Proj { + param($Verify = $null, $Domain = $null, $Deploy = $null) + $p = [pscustomobject]@{} + if ($null -ne $Verify) { $p | Add-Member verify ([pscustomobject]$Verify) } + if ($null -ne $Domain) { $p | Add-Member domain $Domain } + if ($null -ne $Deploy) { $p | Add-Member deploy ([pscustomobject]$Deploy) } + return $p + } +} + +Describe "Get-VerifyAttempts" { + + It "puts the docker-network read first when configured" { + $proj = New-Proj -Verify @{ viaProxy = "evo_edge_proxy"; upstream = "app:80"; port = 8005 } -Domain "x.example" + $attempts = Get-VerifyAttempts -Proj $proj -ExecCmd "docker exec ..." + $attempts[0].Kind | Should -Be 'exec' + ($attempts | ForEach-Object Kind) | Should -Be @('exec', 'port', 'edge') + } + + It "never asks the edge for a project with no domain (the #101 trap)" { + # The shape that hit #101: viaProxy + port, no domain. The old code + # fell back to the bare IP here and read another product's counter. + $proj = New-Proj -Verify @{ viaProxy = "evo_edge_proxy"; upstream = "deploy-app-1:8000"; port = 8005; path = "/health" } + $attempts = Get-VerifyAttempts -Proj $proj -ExecCmd "docker exec ..." + ($attempts | ForEach-Object Kind) | Should -Not -Contain 'edge' + } + + It "returns an empty plan when nothing trustworthy exists" { + # No verify config, no domain: the caller must SKIP, not guess. + $attempts = Get-VerifyAttempts -Proj (New-Proj) -ExecCmd "" + $attempts.Count | Should -Be 0 + } + + It "keeps the edge for a project with a domain, with its Host header" { + $proj = New-Proj -Domain "jwks.example" + $attempts = Get-VerifyAttempts -Proj $proj -ExecCmd "" + $attempts.Count | Should -Be 1 + $attempts[0].Kind | Should -Be 'edge' + $attempts[0].HostHeader | Should -Be "jwks.example" + } + + It "prefers deploy.verifyHost over domain for the edge Host header" { + $proj = New-Proj -Domain "old.example" -Deploy @{ verifyHost = "new.example" } + $attempts = Get-VerifyAttempts -Proj $proj -ExecCmd "" + $attempts[0].HostHeader | Should -Be "new.example" + } + + It "carries the verify path into the port attempt, defaulting sensibly" { + $proj = New-Proj -Verify @{ port = 8005; path = "/health" } + (Get-VerifyAttempts -Proj $proj -ExecCmd "")[0].Path | Should -Be "/health" + $proj2 = New-Proj -Verify @{ port = 9000 } + (Get-VerifyAttempts -Proj $proj2 -ExecCmd "")[0].Path | Should -Be "/api/build-version" + } + + It "survives the PS 5.1 one-element unroll" { + # A single attempt must still come back as something with .Count and + # index access - the pipeline trap that deadlocked ztests day 2. + $proj = New-Proj -Verify @{ port = 8005 } + $attempts = Get-VerifyAttempts -Proj $proj -ExecCmd "" + $attempts.Count | Should -Be 1 + $attempts[0].Kind | Should -Be 'port' + } +} + +Describe "Get-VerifyTimeout" { + + It "uses the caller's default when the project says nothing" { + Get-VerifyTimeout -Proj (New-Proj) -DefaultSec 30 | Should -Be 30 + } + + It "lets a slow-booting project widen its own window" { + # An app that runs database migrations in its entrypoint exceeds + # 30s on every deploy that ships one, and a warning that fires on + # routine success trains people to ignore the real one. + $proj = New-Proj -Verify @{ viaProxy = "p"; upstream = "u"; timeoutSeconds = 120 } + Get-VerifyTimeout -Proj $proj -DefaultSec 30 | Should -Be 120 + } +} diff --git a/zdeploy.ps1 b/zdeploy.ps1 index 8fe8b4e..882aa7c 100644 --- a/zdeploy.ps1 +++ b/zdeploy.ps1 @@ -18,12 +18,11 @@ # zdeploy all -Note "weekly release" # zdeploy ztokens evo # refresh token-stats.json, then ship the site with it # -# "ztokens" is an OPTIONAL pseudo-project, not a zconfig entry: it runs -# `ztokens -Publish` from a sibling ztokens checkout, if you have one, to -# refresh a token-stats.json a site can chart. With no such checkout the step -# prints a skip and the rest of the run is unaffected. `all` runs it first -# automatically; called standalone, list it before a site project (as above) so -# that project's deploy zip picks up the freshly written file. +# "ztokens" is a pseudo-project, not a zconfig entry: it runs `ztokens -Publish` +# from the sibling ztokens repo, refreshing the token-stats.json the public +# zscripts page charts. `all` runs it first automatically; called standalone, +# list it before a site project (as above) so that project's deploy zip picks +# up the freshly written file. # # Flow (python/vite/nextjs): zip source -> free server disk space -> scp up -> # unzip into remote.path (preserving server-side .env* files and anything in @@ -72,7 +71,7 @@ if ($Projects.Count -eq 0) { Write-Host "Usage: zdeploy [ ...] | all | ztokens [-Note `"message`"]" -ForegroundColor Yellow Write-Host " Projects in zconfig.json: $keys" -ForegroundColor Gray Write-Host " 'all' deploys everything (edge kinds first) and stops at the first failure." -ForegroundColor Gray - Write-Host " 'ztokens' refreshes live-usage stats, if a sibling ztokens checkout exists." -ForegroundColor Gray + Write-Host " 'ztokens' refreshes the live-usage stats published to the zscripts page." -ForegroundColor Gray Stop-ZTracking; exit 1 } @@ -132,10 +131,12 @@ function Invoke-Ec2PreflightCleanup { "echo available_mb=`$avail_mb", "if [ `"`$avail_mb`" -lt 1500 ]; then echo 'ERROR: less than 1.5 GB free on /. Grow the root volume or run: sudo docker system prune -af' >&2; exit 11; fi" ) -join '; ' - ssh @SSH_OPTS -i $PEM_KEY $SSH_TARGET $preflightCmd - if ($LASTEXITCODE -ne 0) { - throw "Server pre-flight cleanup failed (exit $LASTEXITCODE). Root volume too full (need ~1.5 GB free, ideally 3+)." - } + # Also through the wrapper (#119): the out-of-space branch above writes its + # ERROR to stderr and exits 11, so a bare ssh would surface a + # NativeCommandError instead of the actionable message below - exactly when + # the operator most needs to be told what to do. -FailHint keeps it. + Invoke-Ec2Step "server pre-flight cleanup" $preflightCmd ` + -FailHint "Root volume too full (need ~1.5 GB free, ideally 3+). Grow it or run: sudo docker system prune -af" } # Post-deploy cleanup: prune build cache and dangling images created during this deploy. @@ -170,30 +171,96 @@ function Invoke-RemoteUnzip { Invoke-Ec2Step "unzip $ZipName" $bash } -# ── Operator-file preservation (issue #2) ──────────────────────────────────── +# ── Operator-file preservation (issue #2, hardened in #107) ────────────────── # Deploys replace the project directory wholesale, which used to destroy every # operator-managed file except ./.env. These helpers preserve all .env* files # at the project root PLUS any paths listed in deploy.preserve (files or # directories), by tarring them to the home dir before the wipe and extracting # them back after the unzip. Server-side copies win over anything shipped in # the zip — the same semantics ./.env always had. +# +# WHY THE ARCHIVE IS TIMESTAMPED AND NEVER DELETED (#107) +# ------------------------------------------------------- +# This used to write one fixed preserve_.tgz, and preserve's FIRST action +# was `rm -f` on it. That is the opposite of safe. The window between +# `sudo rm -rf` on the project directory and the restore step is the only time +# the tarball is the sole copy of the production secrets - and an interrupted +# run (dropped ssh, exit 255) stops exactly there, leaving a perfect backup +# behind. The next run then deleted that backup before doing anything else, +# tarred a directory that no longer had the files, and reported success. +# `2>/dev/null; true` on the tar is what made it silent. +# +# That destroyed civilcode's production deploy/.env on 2026-08-30. The site +# survived only because the running container still held its environment; a +# restart would have made the loss permanent. +# +# So, three independent changes, any one of which would have prevented it: +# +# 1. Each run writes its own preserve__.tgz and restore no +# longer deletes it. Nothing removes an archive that has not been +# superseded - retention below prunes old ones instead. +# 2. Preserve first extracts any earlier archives with `tar -k`, which fills +# in files a previous interrupted run lost WITHOUT overwriting anything +# currently on disk. A hand-repaired .env therefore wins over the stale +# copy in the archive. +# 3. Preserve refuses to continue if it captured nothing while an earlier +# archive for the same key did have contents. Capturing zero files is +# normal for a project with no operator files (edge, gitea, landing) and +# catastrophic for one that has them; the prior archive is what tells the +# difference. +# +# One stamp per zdeploy process, so preserve and restore agree on the filename +# without threading it through every call site. +$script:PreserveStamp = Get-Date -Format 'yyyyMMdd-HHmmss' +$script:PreserveKeep = 5 + +function Get-PreserveTarball { + param([string]$Key) + "$RemoteHome/preserve_${Key}_$($script:PreserveStamp).tgz" +} + function Save-OperatorFiles { param([string]$Key, $Proj, [string]$RemotePath) $paths = @('.env*') if ($Proj.deploy -and $Proj.deploy.preserve) { $paths += @($Proj.deploy.preserve) } $spec = $paths -join ' ' - $tarball = "$RemoteHome/preserve_${Key}.tgz" - # NOTE: no embedded quotes or $( ) here - PowerShell 5.1 strips embedded - # double quotes when passing args to ssh.exe, silently corrupting the - # remote command. Globs expand remotely; tar archives whatever exists - # and its nonzero exit for missing paths is deliberately swallowed. - Invoke-Ec2Step "preserve operator files ($spec)" "rm -f $tarball; cd $RemotePath && tar -czf $tarball $spec 2>/dev/null; true" + $tarball = Get-PreserveTarball -Key $Key + $list = $tarball -replace '\.tgz$', '.list' + $glob = "$RemoteHome/preserve_${Key}_*" + $drop = $script:PreserveKeep + 1 + + # NOTE: no embedded double quotes or $( ) here - PowerShell 5.1 strips + # embedded double quotes when passing args to ssh.exe, silently corrupting + # the remote command, and $( ) would be evaluated locally. Remote shell + # variables are backtick-escaped so PowerShell leaves them alone. Globs + # expand remotely; tar's nonzero exit for missing paths is swallowed, but + # an empty capture is NOT (see the guard below). + $bash = + "cd $RemotePath || exit 9; " + + "for t in ${glob}.tgz; do [ -e `$t ] && tar -xzkf `$t -C $RemotePath 2>/dev/null; done; true; " + + "tar -czf $tarball $spec 2>/dev/null; " + + "tar -tzf $tarball > $list 2>/dev/null; " + + "if [ ! -s $list ]; then " + + "for p in ${glob}.list; do " + + "if [ -s `$p ] && [ `$p != $list ]; then " + + "echo PRESERVE CAPTURED NOTHING BUT AN EARLIER ARCHIVE HAS FILES; exit 8; " + + "fi; " + + "done; " + + "fi; " + + "ls -1t ${glob}.tgz 2>/dev/null | tail -n +$drop | xargs -r rm -f; " + + "ls -1t ${glob}.list 2>/dev/null | tail -n +$drop | xargs -r rm -f; " + + "exit 0" + + Invoke-Ec2Step "preserve operator files ($spec)" $bash ` + -FailHint "Refusing to wipe $RemotePath - see $glob.tgz on the server." } function Restore-OperatorFiles { param([string]$Key, [string]$RemotePath) - $tarball = "$RemoteHome/preserve_${Key}.tgz" - Invoke-Ec2Step "restore operator files" "test -f $tarball && tar -xzf $tarball -C $RemotePath; rm -f $tarball; true" + $tarball = Get-PreserveTarball -Key $Key + # Overwrites, deliberately: server-side operator files beat whatever the + # zip shipped. The archive is left in place - see the header. + Invoke-Ec2Step "restore operator files" "test -f $tarball && tar -xzf $tarball -C $RemotePath; true" } # ── Deploy verification (build-version match, not just HTTP 200 — a 200 can be @@ -247,54 +314,87 @@ function Wait-VerifyStaticBuild { function Wait-VerifyApiBuild { param([string]$Key, $Proj, [string]$ExpectedLabel, [int]$TimeoutSec = 60) Write-Host "`n--- [$Key] Live build verification (expect $ExpectedLabel) ---" -ForegroundColor Cyan - $headers = @{} - # deploy.verifyHost overrides domain for verification only. The Host header - # decides which edge vhost answers, and a project's public host can be - # deliberately unroutable while the app is perfectly healthy - that is why - # the override exists. Reach for it when a domain is being retired ahead of - # its replacement: the old host may be returning 410 while the new one has - # no DNS yet, so neither answers even though the app is fine. - $verifyHost = if ($Proj.deploy -and $Proj.deploy.verifyHost) { $Proj.deploy.verifyHost } else { $Proj.domain } - if ($verifyHost) { $headers['Host'] = $verifyHost } - # Preferred when the project configures it: read the version from a - # container ON the shared docker network rather than through the public - # proxy. A service that publishes no port cannot be curled from the host - # at all, and the proxy answers from whichever vhost matches the Host - # header - so a container with no public route gets another site's - # version back. See Get-ServerSideVersionCommand. - $execCmd = Get-ServerSideVersionCommand -Proj $Proj - $useExec = $false - if ($execCmd) { - $probe = (ssh @SSH_OPTS -i $PEM_KEY $SSH_TARGET $execCmd | Out-String).Trim() - if (Get-LabelFromVersionJson $probe) { $useExec = $true } - } - if ($useExec) { - Write-Host " Asking on server: $($Proj.verify.upstream) (via $($Proj.verify.viaProxy))" -ForegroundColor DarkGray + # deploy.verifyHost (or domain) decides which edge vhost may be asked; + # both are consumed inside Get-VerifyAttempts now, where "no host at all" + # excludes the edge channel entirely rather than defaulting to whatever + # vhost the proxy serves (#101). verifyHost exists for a domain retired + # ahead of its replacement - a takedown once had a project's public host + # answering 410 while the app was healthy; no project sets it today. + # Every retry walks the channels in trust order - docker-network exec, + # then localhost port, then (only with a Host to route by) the edge. + # The choice used to be made ONCE, before the loop, by probing each + # channel - but the probes ran at the exact moment step [5] had + # restarted the app, so both good channels were briefly down and the + # whole window was spent on the edge. For a project with no domain that + # meant the default vhost: one project's check read a DIFFERENT + # project's build number, twice in a single day (#101). Re-resolving per + # retry means the right channel is used the moment the app is back. + # + # The port channel still counts only when the body carries a version: + # one project's verify path is a JWKS endpoint - real, healthy, and no + # version in it - so it falls through to the edge (it has a domain), + # same as it always did. + $execCmd = Get-ServerSideVersionCommand -Proj $Proj + $attempts = Get-VerifyAttempts -Proj $Proj -ExecCmd $execCmd + $TimeoutSec = Get-VerifyTimeout -Proj $Proj -DefaultSec $TimeoutSec + if ($attempts.Count -eq 0) { + # No trustworthy channel exists: no viaProxy, no port, no host to + # route an edge request by. Asking the edge anyway can only reach + # the DEFAULT vhost - a different product - and a check that can + # only ever read someone else's number is worse than no check. + Write-Host " SKIPPED: no way to verify this project without reading the wrong vhost - configure verify.viaProxy/port, or a domain (#101)." -ForegroundColor Yellow + return $false } + Write-Host " Channels, in order: $(($attempts | ForEach-Object { $_.Label }) -join '; ')" -ForegroundColor DarkGray + $sawVersion = $false $deadline = (Get-Date).AddSeconds($TimeoutSec) while ((Get-Date) -lt $deadline) { - try { - if ($useExec) { - $raw = ssh @SSH_OPTS -i $PEM_KEY $SSH_TARGET $execCmd - $live = Get-LabelFromVersionJson ($raw | Out-String) - } else { - $r = Invoke-RestMethod -Uri "http://$EC2_IP/api/build-version" -Headers $headers -TimeoutSec 10 -ErrorAction Stop - $live = if ($r -and $r.build_version) { [string]$r.build_version } else { $null } - } - if ($live) { - if ($live -eq $ExpectedLabel) { - Write-Host " PASS - live build $live matches expected." -ForegroundColor Green - return $true + foreach ($attempt in $attempts) { + $r = $null + try { + switch ($attempt.Kind) { + 'exec' { + $raw = ssh @SSH_OPTS -i $PEM_KEY $SSH_TARGET $execCmd + $r = ($raw | Out-String).Trim() | ConvertFrom-Json -ErrorAction Stop + } + 'port' { + $raw = ssh @SSH_OPTS -i $PEM_KEY $SSH_TARGET "curl -s -m 8 http://localhost:$($attempt.Port)$($attempt.Path)" + $r = ($raw | Out-String).Trim() | ConvertFrom-Json -ErrorAction Stop + } + 'edge' { + $edgeHeaders = @{ 'Host' = $attempt.HostHeader } + $r = Invoke-RestMethod -Uri "http://$EC2_IP/api/build-version" -Headers $edgeHeaders -TimeoutSec 10 -ErrorAction Stop + } } - Write-Host " Live build is $live, expected $ExpectedLabel - waiting..." -ForegroundColor DarkYellow + } catch { + continue # channel not ready; the next one gets its turn } - } catch { - Write-Host " version endpoint not ready yet - waiting..." -ForegroundColor DarkGray + if ($null -eq $r) { continue } + # Two field names in the fleet: some apps answer build_version + # on /api/build-version, others answer version on /health. Both + # are "the build that is live", so accept either rather than + # making every app rename its own field. + $live = if ($r.build_version) { [string]$r.build_version } elseif ($r.version) { [string]$r.version } else { $null } + if (-not $live) { continue } # answered, but not about versions (JWKS etc.) + $sawVersion = $true + if ($live -eq $ExpectedLabel) { + Write-Host " PASS - live build $live matches expected ($($attempt.Label))." -ForegroundColor Green + return $true + } + Write-Host " Live build is $live via $($attempt.Label), expected $ExpectedLabel - waiting..." -ForegroundColor DarkYellow + break # one wrong-version read this pass is enough; retry after the sleep } Start-Sleep -Seconds 3 } - Write-Host " WARNING: live build did not match $ExpectedLabel within ${TimeoutSec}s (a stale build may be cached)." -ForegroundColor Yellow + # Say which failure this actually was: a version that never matched is a + # stale/failed build; channels that never answered is "could not verify", + # and pretending otherwise is how a warning gets ignored. + if ($sawVersion) { + Write-Host " WARNING: live build did not match $ExpectedLabel within ${TimeoutSec}s (upload or Docker build may have failed, or a stale build is cached)." -ForegroundColor Yellow + } else { + Write-Host " WARNING: could not verify within ${TimeoutSec}s - no channel answered with a version (app may still be starting; raise verify.timeoutSeconds if this project boots slowly)." -ForegroundColor Yellow + } return $false } @@ -435,7 +535,7 @@ function Invoke-PythonDeploy { # in the container, is written to .build_version below, and is # proven by the /api/build-version check — which is the thing that # actually establishes what is deployed. - ssh @SSH_OPTS -i $PEM_KEY $SSH_TARGET "echo '$BuildVersion' | sudo tee $remotePath/.build_version > /dev/null" + Invoke-Ec2Step "record the live build number" "echo '$BuildVersion' | sudo tee $remotePath/.build_version > /dev/null" $changelogTool = Join-Path $root "scripts\build_changelog_tool.py" if (Test-Path -LiteralPath $changelogTool) { if ([string]::IsNullOrWhiteSpace($ChangeNote)) { $ChangeNote = "Build deployed" } @@ -443,8 +543,18 @@ function Invoke-PythonDeploy { } Write-Host "`n--- [5] Restarting app to pick up new version ---" -ForegroundColor Cyan - ssh @SSH_OPTS -i $PEM_KEY $SSH_TARGET "cd $composeDir && sudo COMPOSE_BAKE=false docker compose restart $appSvc" - if ($LASTEXITCODE -ne 0) { throw "App restart after build bump failed (exit $LASTEXITCODE)" } + # Through Invoke-Ec2Step, not a bare ssh: `docker compose restart` + # writes " Container Restarting" to STDERR as ordinary + # progress, and under ErrorActionPreference='Stop' PS 5.1 turns any + # native stderr line into a terminating NativeCommandError whatever + # the exit code. That threw here on a deploy that had fully + # succeeded - and it threw BEFORE Wait-VerifyApiBuild, so the step + # that proves what is actually deployed never ran (#119). The + # wrapper flattens stderr and judges by exit code alone, and throws + # on non-zero itself, so the hand-written check is gone with it. + Invoke-Ec2Step "restart $appSvc to pick up the new build" ` + "cd $composeDir && sudo COMPOSE_BAKE=false docker compose restart $appSvc" ` + -FailHint "App restart after the build bump failed." Wait-VerifyApiBuild -Key $Key -Proj $Proj -ExpectedLabel $BuildVersion -TimeoutSec 30 | Out-Null } elseif ($Proj.verify -and $Proj.verify.port) { @@ -646,7 +756,7 @@ function Invoke-NextDeploy { # offline for the whole build - minutes for a Next.js app - and left it # offline if the build failed. That is not hypothetical: one deploy # stopped the stack, the build did not finish, and the site served 502 - # for 19 hours with no container running at all. The + # for 19 hours with no container at all. The # old image keeps serving while the new one builds, so a failed build is # now harmless and the outage is the seconds between down and up. # @@ -670,7 +780,7 @@ function Invoke-NextDeploy { # here: a compose stack behind the edge proxy usually publishes to # 127.0.0.1 only, so that probe can never answer and the old "is the port # open in the security group?" warning sent you chasing a firewall rule - # for an app that was already up. + # for an app that was already up. See issue #28. if ($Proj.verify -and $Proj.verify.port) { Test-DeployHealth -Key $Key -Proj $Proj -TimeoutSec 60 | Out-Null } elseif ($Proj.domain) { @@ -752,10 +862,11 @@ function Invoke-EdgeDeploy { } # Content subdirectories the proxy serves (fonts/, vendor/, ...) ship too — - # only server-side state stays put. Skipping them is how self-hosted assets - # silently never reach prod: docker creates empty mount-point dirs and nginx - # serves 404s from them, so fonts fall back and vendored JS never loads. - $skipDirs = @('nginx-logs', '.git') + # only server-side state stays put. Skipping them is how the self-hosted + # Chart.js and fonts silently never reached prod (charts rendered blank). + # .pytest_cache is a local test artifact, already gitignored; it has no + # business on the proxy box and only adds noise to the upload log. + $skipDirs = @('nginx-logs', '.git', '.pytest_cache') $dirs = @(Get-ChildItem -LiteralPath $root -Directory | Where-Object { $skipDirs -notcontains $_.Name }) foreach ($d in $dirs) { Write-Host " >> uploading $($d.Name)/ (recursive)" -ForegroundColor DarkCyan @@ -781,9 +892,9 @@ function Invoke-EdgeDeploy { function Invoke-StaticDeploy { param([string]$Key, $Proj) - # Plain static sites - no build, no container of their own. A shared web - # container serves them straight off disk, so shipping the files IS the - # deploy: there is nothing to restart afterwards. + # Plain static sites - no build, no container of their own. The landing + # container serves them straight off disk out of /srv/$host, so shipping + # the files IS the deploy: there is nothing to restart afterwards. $root = Join-Path $Proj.localRoot $Proj.siteDir $remotePath = $Proj.remote.path @@ -801,11 +912,8 @@ function Invoke-StaticDeploy { } # A large media file uploaded in place is served half-written to anyone who - # requests it mid-copy. Ship each directory to a sibling, then swap it in - - # the swap is a rename, so the switch is atomic and visitors never see a - # partial file. - $skipDirs = @($script:JunkDirNames) + @('.github') - if ($Proj.deploy -and $Proj.deploy.skipDirs) { $skipDirs += @($Proj.deploy.skipDirs) } + # requests it mid-copy. Ship the directory to a sibling, then swap it in. + $skipDirs = @('.git', '.pytest_cache', 'node_modules') $dirs = @(Get-ChildItem -LiteralPath $root -Directory | Where-Object { $skipDirs -notcontains $_.Name }) foreach ($d in $dirs) { Write-Host " >> uploading $($d.Name)/ (recursive, staged)" -ForegroundColor DarkCyan @@ -870,11 +978,10 @@ function Invoke-DockerDeploy { Write-DeployLocation -Proj $Proj } -# Pseudo-project "ztokens": not a zconfig entry, no compose stack, and entirely -# optional. Runs `ztokens -Publish` from a sibling ztokens checkout so a site -# that charts usage data has something current to ship. Missing checkout, or a -# failure, warns rather than aborting the rest of the deploy list - it is a -# nice-to-have refresh, not a deploy step. +# Pseudo-project "ztokens": not a zconfig entry, no compose stack. Runs +# `ztokens -Publish` from the sibling ztokens repo so the public zscripts page +# has current data. A failure here warns rather than aborting the rest of the +# deploy list - it's a nice-to-have refresh, not a deploy step. function Invoke-ZTokensPublish { Write-Host "`n=== ztokens: refreshing live-usage stats ===" -ForegroundColor Cyan $ztokensScript = Join-Path (Split-Path -Parent $PSScriptRoot) "ztokens\ztokens.ps1"