Compare commits
41 Commits
a29e0708e8
...
main
| Author | SHA1 | Date | |
|---|---|---|---|
| 8f7e409ec4 | |||
| f4b2c15a9f | |||
| 2a0a793f19 | |||
|
|
aeeb26f4e9 | ||
| 9c963514f1 | |||
| 565cecfb50 | |||
| 1752a95408 | |||
| 1dc9d38c80 | |||
| 6a005dae3b | |||
| de5b1b7ea8 | |||
| 6dbc83a449 | |||
| 730ebaff2f | |||
| dd17021402 | |||
| 19feac6d57 | |||
| 809a13eebe | |||
| 004b397b94 | |||
| 29e30693ea | |||
| 3f7f6c988d | |||
| 3e864aa919 | |||
| c3fe4422f2 | |||
| 99b1988504 | |||
| 5219bd5edb | |||
| f7910bf42b | |||
| 24aeadde83 | |||
| 5391f50755 | |||
| fed5d92034 | |||
| 7b70b7edc6 | |||
| 74e246e67a | |||
| e05f8f1fca | |||
| b102ab8de7 | |||
| 2f3e9c2634 | |||
| 86d051da48 | |||
| 6ce24586bd | |||
| 7242b09e3a | |||
| fbf47980d9 | |||
| b9238040a6 | |||
| aba696df79 | |||
| 974679a432 | |||
| 26f99265ca | |||
| a9df70cde0 | |||
| e0426ecb01 |
7
Makefile
7
Makefile
@@ -42,7 +42,7 @@ $(eval $(ARGS):;@:)
|
||||
endif
|
||||
|
||||
.DEFAULT_GOAL := help
|
||||
.PHONY: help build start stop dist docs cluster deploy component
|
||||
.PHONY: help build start stop dist theme docs cluster deploy component
|
||||
|
||||
help: ## list targets
|
||||
@grep -hE '^[a-z]+:.*?##' $(MAKEFILE_LIST) | sed 's/:.*##/\t/' | expand -t16
|
||||
@@ -61,6 +61,11 @@ stop: ## stop a running room [<room>]
|
||||
dist: ## compile the plexus UIs to single files [<room>]
|
||||
bash ctrl/dist.sh $(or $(ARGS),$(ROOM))
|
||||
|
||||
# ── theme ──────────────────────────────────────────────────────────────────
|
||||
|
||||
theme: ## ad-hoc pages [new|parts|bake|check|export|run FILE]
|
||||
bash ctrl/theme.sh $(or $(ARGS),bake)
|
||||
|
||||
# ── docs ───────────────────────────────────────────────────────────────────
|
||||
|
||||
docs: ## documentation [serve [port]|graphs [theme]] (default serve)
|
||||
|
||||
21
berth/.gitignore
vendored
Normal file
21
berth/.gitignore
vendored
Normal file
@@ -0,0 +1,21 @@
|
||||
# Machine-local config and credentials. Never committed.
|
||||
ctrl/.env
|
||||
|
||||
# Rendered output. Regenerate with `make services render <target>`.
|
||||
# Generated config is an artifact, not source — the estate file is the source.
|
||||
#
|
||||
# The path is ctrl/render/out/, NOT render/out/. A pattern containing a slash is
|
||||
# anchored to the directory holding this .gitignore, so `render/out/` would mean
|
||||
# berth/render/out/ — which does not exist, and the real output would have been
|
||||
# committed. Caught by `git check-ignore -v`, which is the only way to be sure.
|
||||
ctrl/render/out/
|
||||
|
||||
# The "default" scratch bucket: always gitignored, never versioned.
|
||||
def/
|
||||
|
||||
# Key material. NEVER committed.
|
||||
#
|
||||
# Anchored to ctrl/ for the same reason as render/out/ above: a pattern with a
|
||||
# slash resolves against this file's own directory. `git check-ignore -v` is the
|
||||
# only way to confirm it, and ctrl/vpn.sh refuses to write a key until it does.
|
||||
ctrl/.secrets/
|
||||
78
berth/Makefile
Normal file
78
berth/Makefile
Normal file
@@ -0,0 +1,78 @@
|
||||
# One target per ctrl/ script; the subcommand is an argument, not a second
|
||||
# target: `make estate show`, not `make estate-show`. The logic lives in the
|
||||
# scripts, never here.
|
||||
#
|
||||
# Config layers, weakest first: ctrl/versions.env < ctrl/env.d/<target>.env <
|
||||
# ctrl/.env < the environment. So `make estate plan TARGET=gcp` beats all.
|
||||
#
|
||||
# Every target defaults to its READ-ONLY verb, and the verbs that change a live
|
||||
# estate are not reachable by a bare word. Rationale: README.md.
|
||||
|
||||
ESTATE := $(or $(shell sed -n 's/^ESTATE=//p' ctrl/.env 2>/dev/null),$(shell ls estate/*.json 2>/dev/null | head -1 | xargs -r basename | sed 's/\.json$$//'))
|
||||
TARGET := $(or $(shell sed -n 's/^TARGET=//p' ctrl/.env 2>/dev/null),aws)
|
||||
|
||||
.PHONY: help check selftest estate services vpn dns certs host ports registry docs
|
||||
|
||||
help: ## list targets
|
||||
@grep -hE '^[a-z][a-z-]*:.*?##' $(MAKEFILE_LIST) | sed 's/:.*##/\t/' | expand -t16
|
||||
|
||||
# ── preflight ──────────────────────────────────────────────────────────────
|
||||
|
||||
check: ## is this estate coherent? reports, never fixes
|
||||
bash ctrl/check.sh
|
||||
|
||||
selftest: ## does berth still do what it says? exits 1 if not
|
||||
bash ctrl/selftest.sh
|
||||
|
||||
ports: ## port map [show|verify] (default show)
|
||||
bash ctrl/ports.sh $(or $(ARGS),show)
|
||||
|
||||
# ── the estate ─────────────────────────────────────────────────────────────
|
||||
|
||||
estate: ## the estate [show|list|plan|apply|destroy] (default show)
|
||||
bash ctrl/estate.sh $(or $(ARGS),show)
|
||||
|
||||
services: ## gateway routes [list|render <target>|deploy] (default list)
|
||||
bash ctrl/services.sh $(or $(ARGS),list)
|
||||
|
||||
# ── the network ────────────────────────────────────────────────────────────
|
||||
|
||||
vpn: ## overlays [list|show <ov>|check|render|keygen] (default list)
|
||||
bash ctrl/vpn.sh $(or $(ARGS),list)
|
||||
|
||||
# ── names and trust ────────────────────────────────────────────────────────
|
||||
|
||||
dns: ## DNS records [list|add|add-wildcard|remove] (default list)
|
||||
bash ctrl/dns.sh $(or $(ARGS),list)
|
||||
|
||||
certs: ## TLS [status|verify|renew|push] (default status)
|
||||
bash ctrl/certs.sh $(or $(ARGS),status)
|
||||
|
||||
# ── the box ────────────────────────────────────────────────────────────────
|
||||
|
||||
host: ## the remote box [status|ports|services] (default status)
|
||||
bash ctrl/host.sh $(or $(ARGS),status)
|
||||
|
||||
registry: ## the image registry [status] (default status)
|
||||
bash ctrl/registry.sh $(or $(ARGS),status)
|
||||
|
||||
# ── docs ───────────────────────────────────────────────────────────────────
|
||||
|
||||
docs: ## documentation [serve|graphs] (default serve)
|
||||
bash ctrl/docs.sh $(or $(ARGS),serve)
|
||||
|
||||
# ── swallowing the argument words — MUST BE LAST IN THIS FILE ──────────────
|
||||
#
|
||||
# Words after the target are arguments, but make reads each as a goal, so each
|
||||
# gets a no-op rule. This block must come AFTER the real targets: when an
|
||||
# argument names one (`make host ports`, `make vpn check`), the last definition
|
||||
# wins, and it has to be the no-op. With it first, make ran both scripts.
|
||||
#
|
||||
# Make's "overriding recipe" warning is the swallow working as intended.
|
||||
ARGS := $(wordlist 2,$(words $(MAKECMDGOALS)),$(MAKECMDGOALS))
|
||||
ifneq ($(ARGS),)
|
||||
$(eval $(ARGS):;@:)
|
||||
# .PHONY too: some of those words name real directories (ctrl, estate, render),
|
||||
# and make treats an existing directory as already built.
|
||||
.PHONY: $(ARGS)
|
||||
endif
|
||||
250
berth/README.md
Normal file
250
berth/README.md
Normal file
@@ -0,0 +1,250 @@
|
||||
# berth
|
||||
|
||||
spr's deploy half. A rig is a mobile installation; a **berth** is the allocated, paid-for
|
||||
place where it is moored and actually operates. Local rig → remote berth.
|
||||
|
||||
```bash
|
||||
make check # is this estate coherent? reports, never fixes
|
||||
make estate show # the description, resolved
|
||||
make vpn check # the overlay: addresses, routing, bindings, key hygiene
|
||||
make services render aws
|
||||
make ports verify # does the local map still agree with rig?
|
||||
```
|
||||
|
||||
`rm -rf berth/` is the uninstall.
|
||||
|
||||
---
|
||||
|
||||
## What berth is
|
||||
|
||||
**One description of an estate, with a swappable executor.** The description is the artifact;
|
||||
the tool that runs it is a rendering.
|
||||
|
||||
| layer | what berth uses | note |
|
||||
| --- | --- | --- |
|
||||
| overlay | **WireGuard**, config generated per peer | GPL-2.0, in-kernel, and **no coordination server** |
|
||||
| infra | **OpenTofu** — one executor, not a pair | plain `.tf`; `terraform` works identically |
|
||||
| gateway | Caddy locally, nginx on the box | a projection with per-target rules, not a format conversion |
|
||||
| pipeline | Woodpecker | Actions/GitLab reachable from the same description; not built |
|
||||
|
||||
The estate's facts — domain, hosts, ports, instance size, firewall rules, services — live in
|
||||
**one file every rendering reads** (`estate/<name>.json`), rather than being restated in
|
||||
one tool's language and again in another's. **Swappability is bought by the description, not
|
||||
by maintaining two renderings** — a rendering you *can* produce, not one you *must* keep in
|
||||
step. Two live renderings cost you every resource twice, forever, with nothing enforcing that
|
||||
they agree.
|
||||
|
||||
**The seam belongs in a script, not in a tool.** A tool's schema is a ceiling you do not
|
||||
control.
|
||||
|
||||
*(This once leaned on a specific precedent, since withdrawn — `✖ B2` in [STALE.md](STALE.md).)*
|
||||
|
||||
**OpenTofu, and only OpenTofu.** Terraform has been BUSL-licensed since 2023; OpenTofu is the
|
||||
CLI-compatible MPL-2.0 fork (Linux Foundation). Write plain `.tf` that runs under both; the
|
||||
scripts say `tofu`, and `terraform` works identically.
|
||||
|
||||
**Pipelines are the second executor axis, and are not built.** Woodpecker is the
|
||||
self-hosted rendering; Actions and GitLab CI are the standards that must be reachable from
|
||||
the same description. Named here so the infra seam is not designed in a way that forecloses
|
||||
it.
|
||||
|
||||
---
|
||||
|
||||
## The overlay is berth's network layer
|
||||
|
||||
Two instances in different clouds cannot share a VPC. A WireGuard overlay gives them one flat
|
||||
address space that berth owns and can reproduce on any provider — and, under a flaky
|
||||
environment, a second layer beneath whatever the provider offers.
|
||||
|
||||
That inverts the usual cloud pattern. Instead of a VPC with security groups and private
|
||||
subnets, each instance gets a public IP, opens **only** the WireGuard port, and carries
|
||||
everything else inside the tunnel. **The security boundary moves out of the provider's VPC
|
||||
and into a layer that is identical on AWS, on GCP, and on a laptop behind NAT.**
|
||||
|
||||
**Plain WireGuard, not Tailscale/Headscale/NetBird.** Tailscale's client is open but its
|
||||
coordination plane is proprietary SaaS; Headscale and NetBird are open but add a control plane
|
||||
to run. Plain WireGuard needs no server at all.
|
||||
|
||||
**The cost is real and accepted:** no NAT traversal, no relay, no peer discovery. A peer behind
|
||||
NAT must dial one with a public endpoint. Fine here — the instances have public IPs and the dev
|
||||
box roams — but two roaming peers cannot reach each other. That is why `PersistentKeepalive` is
|
||||
a checked invariant rather than a detail.
|
||||
|
||||
**Keys never enter the description.** Private keys are generated on the peer that owns them
|
||||
(`make vpn keygen`) into `ctrl/.secrets/` and injected only at render time; public keys live in
|
||||
the estate, because a config cannot be built without them. `vpn.sh` **refuses to write** either
|
||||
a key or a rendered config until `git check-ignore` confirms the path is ignored — this repo
|
||||
has already been bitten once by a `.gitignore` pattern anchoring to the wrong directory.
|
||||
|
||||
This also decides something about the IaC layer: **OpenTofu must never generate a WireGuard
|
||||
private key**, because Terraform-lineage state stores every resource attribute in plaintext.
|
||||
|
||||
---
|
||||
|
||||
## Two rules that are not style preferences
|
||||
|
||||
### 1. Every default is the read-only verb
|
||||
|
||||
```
|
||||
make estate -> show make certs -> status
|
||||
make dns -> list make host -> status
|
||||
make estate apply / destroy -> print the plan, then refuse without --yes
|
||||
```
|
||||
|
||||
rig's `make cluster` defaults to `up`, because every rig verb is safe — a kind cluster is
|
||||
disposable. berth's are not: `tofu destroy` costs money and takes live DNS with it. **A
|
||||
tool where every verb is safe must not grow verbs that are not.**
|
||||
|
||||
The failure this prevents is not hypothetical. `ppl/ctrl/certs.sh:42` is `CMD="${1:-all}"`,
|
||||
so a bare `./ctrl/certs.sh` there issues a real Let's Encrypt cert, rsyncs it to the gateway,
|
||||
and reloads nginx. berth's `certs` defaults to `status`.
|
||||
|
||||
### 2. berth and rig share a convention, not code
|
||||
|
||||
Neither imports the other. They match on **shape** — the `make <noun> <verb>` dispatch, the
|
||||
four-source config layering, the key names — and consistency is verified by
|
||||
**recomputation**: `make ports verify` recomputes rig's `20000 + (cksum(name) % 200) * 10`
|
||||
to check the local map, rather than sourcing rig's `lib/config.sh`.
|
||||
|
||||
Copying three stable lines is the whole cost of not coupling them. A shared library would
|
||||
put something outside `rig/` on rig's path, and rig's promise is that
|
||||
`grep -rIn -iE 'soleprint|\bspr\b'` across it returns nothing.
|
||||
|
||||
**rig is also unaware that berth exists.** `ppl/local/Caddyfile` is berth's to generate; rig
|
||||
must not reference `local.ar` — its handover scrub refuses the string.
|
||||
|
||||
---
|
||||
|
||||
## The gateway doctrine
|
||||
|
||||
Practised across this codebase for a long time and never written down, so: written down.
|
||||
|
||||
- **Caddy where routing is dynamic and config-driven** — the in-cluster gateway that
|
||||
multiplexes by Host header, and the host-side `.local.ar` name→port map.
|
||||
- **nginx where it is a static server or a plain long-running compose service on the box.**
|
||||
- **Envoy in `mpr`** — a deliberate one-off, not a third pattern.
|
||||
- **`ingress-nginx` only as a kind addon** — a different thing again from either gateway.
|
||||
|
||||
### Rendering is a projection, not a format conversion
|
||||
|
||||
The local Caddyfile and the box's nginx are not two spellings of the same content:
|
||||
|
||||
| | local (Caddy) | cloud (nginx) |
|
||||
| --- | --- | --- |
|
||||
| granularity | one file | one file per vhost |
|
||||
| blocks per service | one | two (`:80` redirect + `:443` server) |
|
||||
| TLS | none; every address needs an explicit `:80` | one shared wildcard cert |
|
||||
| upstream | `localhost:<port>` | container name + `resolver 127.0.0.11` |
|
||||
| name depth | free | constrained by the cert |
|
||||
| ambiguity | most specific wins | exact, else `default_server` (= load order) |
|
||||
|
||||
The `:80` is not decoration: without it Caddy 2 defaults each site to `:443` with auto-HTTPS,
|
||||
which on `*.local.ar` means cert provisioning that fails and breaks the listener. The
|
||||
variable upstream is not decoration either: naming the upstream in a variable forces runtime
|
||||
DNS resolution, so nginx **starts when the upstream container is absent** — which is what
|
||||
lets one nginx front a dozen independent compose stacks.
|
||||
|
||||
Because the two disambiguate by **opposite** rules, a name set that is unambiguous locally
|
||||
can be ambiguous on the box. `make check` asserts against the projection, not the source.
|
||||
|
||||
### Installing generated vhosts — an order that is not optional
|
||||
|
||||
1. Generated config lands in `conf.d/generated/`, **not** `conf.d/`. `ppl/ctrl/deploy.sh`
|
||||
rsyncs with `--delete`; sharing a directory means one set gets erased.
|
||||
2. `nginx.conf` needs a **third** include line — its `conf.d/*.conf` glob does not recurse,
|
||||
which is why `conf.d/soleprint/*.conf` already needs its own.
|
||||
3. That include changes **load order**, and load order decides which `:443` block catches
|
||||
unmatched names. So `default.conf`'s commented-out `:443 default_server` must be restored
|
||||
**first**. `make check` fails on it deliberately: it is a gate, not a warning.
|
||||
|
||||
---
|
||||
|
||||
## What the checks assert, and why
|
||||
|
||||
The scripts are deliberately thin on comment — the reasoning lives here. Every check below
|
||||
corresponds to something that is wrong, or was wrong, in a real estate.
|
||||
|
||||
### `make check` — the estate
|
||||
|
||||
| assertion | the failure it catches |
|
||||
| --- | --- |
|
||||
| every service name is covered by an issued cert SAN | **a wildcard matches exactly one label.** `*.d.com` covers `a.d.com` but not `a.b.d.com`, which needs its own SAN. Surfaces otherwise as a browser TLS warning, far from its cause |
|
||||
| a `:443 default_server` exists | DNS and the cert are wildcard but nginx matches `server_name` exactly, so without one the fallback for any unknown name is whichever vhost loads first — alphabetically, by accident |
|
||||
| firewall rules and listeners agree | a rule allowing a port nothing listens on is **dead config**; a service no compose file declares is **undocumented state**. Neither is visible from one side alone, which is why the inventory has two halves |
|
||||
| `HOST` is an ssh alias, never a hostname | there is no `Host <domain>` block, so a bare hostname falls through to the global defaults, ssh offers every key in the agent in turn, and `MaxAuthTries` (6) trips with *"Too many authentication failures"* before reaching the right one. The aliases set `IdentitiesOnly yes` |
|
||||
|
||||
### `make vpn check` — the overlay
|
||||
|
||||
| assertion | the failure it catches |
|
||||
| --- | --- |
|
||||
| peer addresses unique and inside the subnet | a duplicate is a silent misroute, never an error |
|
||||
| AllowedIPs do not overlap | AllowedIPs is **cryptokey routing** — the route table and the ACL at once. Overlapping ranges resolve to the last match, so an overlap is both a misroute and an unintended grant |
|
||||
| something carries `PersistentKeepalive` if anything roams | without it a NAT mapping expires and the tunnel works only while traffic flows outward — *"works sometimes"*, the hardest failure to read. Note it is **not** a property of the roaming peer's own entry: the roaming machine sets it on the entry for the peer it **dials** |
|
||||
| the listen port is in the firewall | otherwise no peer can be dialed at all |
|
||||
| no private key in the description | public keys are *also* 44-char base64, so the shape proves nothing. The real assertions are **no field named `priv*`** and **no key outside a `public_key` field** |
|
||||
| overlay-reached services bind a reachable address | **a tunnel cannot reach loopback.** A service on `127.0.0.1` is unreachable over the overlay; one on `0.0.0.0` is reachable but also exposed to the whole LAN |
|
||||
|
||||
### Capturing the overlay
|
||||
|
||||
```bash
|
||||
sudo wg show | make vpn capture --write
|
||||
```
|
||||
|
||||
`wg show` has three forms and **only the first is safe**:
|
||||
|
||||
| form | safe | why |
|
||||
| --- | --- | --- |
|
||||
| `wg show` | **yes** | prints `private key: (hidden)` |
|
||||
| `wg show <if> dump` | **no** | field 1 of the first line *is* the private key |
|
||||
| `wg showconf <if>` | **no** | prints `PrivateKey=` outright |
|
||||
|
||||
`capture` refuses the latter two by shape. It matches peers by **allowed-ips address, not
|
||||
public key** — the keys are exactly what is missing at that point — and **drops a roaming
|
||||
peer's endpoint in the parser**, since that value is a home ISP address and a roaming peer
|
||||
has no stable endpoint anyway.
|
||||
|
||||
## Layout
|
||||
|
||||
```
|
||||
berth/
|
||||
├── Makefile one target per ctrl/ script; the verb is an argument
|
||||
├── STALE.md withdrawn assumptions, each with a check that runs
|
||||
├── estate/<name>.json THE ARTIFACT — one description, many renderings
|
||||
└── ctrl/
|
||||
├── check.sh reports and instructs; never fixes
|
||||
├── estate.sh show | list | plan | apply --yes | destroy --yes
|
||||
├── services.sh list | render <aws|gcp|local> | deploy
|
||||
├── ports.sh show | verify (the rig coincidence check)
|
||||
├── dns.sh certs.sh host.sh registry.sh docs.sh
|
||||
├── versions.env pinned toolchain (weakest layer)
|
||||
├── env.d/<target>.env provider shape: aws | gcp
|
||||
├── .env.example -> ctrl/.env, machine-local (gitignored)
|
||||
├── lib/config.sh the four-layer load, from rig
|
||||
├── lib/estate.sh reading and projecting the estate
|
||||
└── render/*.tmpl nginx vhost shapes, substituted with sed
|
||||
```
|
||||
|
||||
Config layers, weakest first: `versions.env` → `env.d/<target>.env` → `ctrl/.env` → the
|
||||
caller's environment. So `make estate plan TARGET=gcp` beats everything.
|
||||
|
||||
**Identity is explicit — the inversion of rig.** rig derives its name from its folder so that
|
||||
copies never collide. berth refuses to guess, because a deployment has exactly one production
|
||||
and a wrong guess acts on the wrong estate. `ESTATE` names a file; the only convenience is
|
||||
that a single `estate/*.json` is used without being asked for.
|
||||
|
||||
**`python3`, not `jq`.** rig ships a pinned static `jq` because its floor is "docker and
|
||||
nothing else" on a machine it does not control. berth's floor is already higher, so
|
||||
`python3` is a dependency it *has* rather than one it *adds* — the same reasoning by which
|
||||
rig chose `sed` over `envsubst`.
|
||||
|
||||
---
|
||||
|
||||
## Status
|
||||
|
||||
**B0 (this) is the shape.** `estate/mcrn.json` is marked `UNVERIFIED`: it records what the
|
||||
repos *claim*, because `ppl/infra/` was written and never applied — no `~/.pulumi`, no
|
||||
`venv`, no stack state, files dated `mar 6`. B1's read-only inventory is what replaces those
|
||||
claims with observations. Until then, `estate plan` and `estate apply` refuse: there is
|
||||
nothing truthful to compare against yet.
|
||||
|
||||
`make check` currently fails on two real things — see `def/plans/36.0/berth.md`.
|
||||
104
berth/STALE.md
Normal file
104
berth/STALE.md
Normal file
@@ -0,0 +1,104 @@
|
||||
# berth — withdrawn assumptions
|
||||
|
||||
**Everything in this file is no longer true.**
|
||||
|
||||
It exists so the live docs stay short and so a withdrawn assumption cannot quietly return:
|
||||
each entry carries a **check**, and `ctrl/selftest.sh` runs every one of them. A retraction
|
||||
that is only prose is a retraction nobody re-reads.
|
||||
|
||||
Kept rather than deleted for the reason the docgen thread already wrote down —
|
||||
*a requirement that disappears without explanation comes back.*
|
||||
|
||||
### For agents
|
||||
|
||||
- **Do not restate these** in a plan, a README or a comment. One line pointing here is enough.
|
||||
- **Ids are stable.** Cite `✖ B3`; do not re-explain it.
|
||||
- **When you withdraw an assumption, move it here** — quoted claim, where it came from, what
|
||||
superseded it and why, what changed in the code, and a check that proves it is gone.
|
||||
- **Only withdrawn things belong here.** A warning that is still actionable is a live rule,
|
||||
however historical it sounds, and stays where it is.
|
||||
|
||||
---
|
||||
|
||||
**✖ B1 — "Pulumi is the source in spr, and Terraform must match the config."**
|
||||
*(INDEX §5, carried from 35.3)* Withdrawn 2026-09-12. berth uses **OpenTofu and only
|
||||
OpenTofu**. The rule assumed *open source* and *industry standard* pull apart — Terraform
|
||||
being BUSL, Pulumi being the open alternative. OpenTofu is both: the standard language, plain
|
||||
Terraform-compatible HCL, under MPL-2.0. Swappability was never bought by keeping two
|
||||
renderings; it is bought by `estate/*.json` being the artifact — a rendering you *can*
|
||||
produce, not one you *must* maintain.
|
||||
**Gone from:** `ctrl/versions.env` (no `PULUMI_VERSION`), `ctrl/estate.sh` (`plan` runs one
|
||||
executor), `ctrl/lib/config.sh` (`PULUMI_STACK` → `TOFU_WORKSPACE`), `ctrl/check.sh`
|
||||
(toolchain list).
|
||||
**Deliberately kept:** `README.md` and `estate/mcrn.json` both record that `ppl/infra/` was
|
||||
written and never applied — *"no `~/.pulumi`, no venv, no stack state"*. That is a historical
|
||||
fact about the estate, not a live dependency.
|
||||
**Check:** no `pulumi` in `ctrl/` or `Makefile`.
|
||||
|
||||
**✖ B2 — "ctlptl's rejection is the precedent for *the seam belongs in a script, not a tool*."**
|
||||
*(INDEX §5, `berth/README.md`)* Withdrawn 2026-09-12. **ctlptl was reinstated** — pinned in
|
||||
`rig/ctrl/versions.env` at v0.9.4 — and had been removed for the wrong reason. The rule may
|
||||
still hold; it now has to stand on its own reasoning rather than that example.
|
||||
**Gone from:** `README.md` — the argument is stated directly, with no borrowed evidence.
|
||||
**Check:** `ctlptl` appears nowhere in berth.
|
||||
|
||||
**✖ B3 — "`wg show` is safe; `wg showconf` is not."** *(my own note, 2026-09-12)* Incomplete,
|
||||
and the gap is the dangerous one. There are **three** forms, and `wg show <if> dump` puts the
|
||||
**private key in field 1 of the first line**. Stated as a two-way distinction, the `dump` form
|
||||
reads as safe.
|
||||
**Now:** `wg show` plain is safe; `dump` and `showconf` are not. `ctrl/vpn.sh capture` refuses
|
||||
the latter two **by shape**, rather than parsing around them.
|
||||
**Check:** `vpn.sh` names all three forms, and `capture` rejects both unsafe ones.
|
||||
|
||||
**✖ B4 — "A roaming peer must set `PersistentKeepalive` on its own entry."**
|
||||
*(`ctrl/vpn.sh`, first draft)* Wrong side, and wrong in the direction that looks fine:
|
||||
`PersistentKeepalive` is set per-peer in a config, so the roaming machine sets it on the entry
|
||||
for the peer it **dials**. The original check would have **warned on a correctly configured
|
||||
overlay**.
|
||||
**Now:** checked once per overlay — if anything roams, some peer entry must carry a keepalive.
|
||||
**Check:** the invariant is not keyed on the roaming peer's own `keepalive` field.
|
||||
|
||||
**✖ B5 — "The Makefile's pass-through block goes near the top, with the other variables."**
|
||||
*(`Makefile`, inherited from rig's layout)* Withdrawn 2026-09-12. When a subcommand **names a
|
||||
real target**, make has two recipes for it and the **last definition wins** — so with the block
|
||||
first, `make host ports` ran `ctrl/host.sh ports` *and* `ctrl/ports.sh ports`, the second
|
||||
failing because `ports` is not one of its verbs. Same for `make host services`, `make vpn
|
||||
check`, `make vpn show estate`.
|
||||
**Now:** the `$(eval $(ARGS):;@:)` block is **last in the file**, so the no-op wins and the
|
||||
word is swallowed — which is what an argument is. Make's *"overriding recipe"* warning is the
|
||||
swallow working.
|
||||
**Check:** every colliding invocation dispatches to exactly one script.
|
||||
|
||||
**✖ B6 — "A base64 key in the description can be caught by its shape."**
|
||||
*(`ctrl/vpn.sh check`, first draft)* WireGuard **public** keys are also 44-char base64 and
|
||||
legitimately live in the estate, so shape alone proves nothing and would flag correct data.
|
||||
**Now:** two assertions instead — **no field named `priv*`**, and **no base64 key outside a
|
||||
`public_key` field**.
|
||||
**Check:** the estate's public keys do not trip the secret check.
|
||||
|
||||
**✖ B7 — "`network.wireguard` is where the overlay is described."** *(`estate/mcrn.json`)*
|
||||
Superseded 2026-09-12: WireGuard is berth's network layer, not one service's transport, so it
|
||||
is a top-level `vpn` block with named overlays and peers. `ctrl/check.sh` and
|
||||
`ctrl/registry.sh` were repointed.
|
||||
**Gone from:** the estate schema — a `wireguard_moved` tombstone marks the old key.
|
||||
**Check:** nothing reads `network.wireguard.*`.
|
||||
|
||||
**✖ B8 — "berth and rig are related through their first uses."** *(early framing)* Withdrawn:
|
||||
**rig and berth are peers — neither depends on the other.** They match on shape (dispatch,
|
||||
config layering, key names) and consistency is verified by **recomputation**, never by
|
||||
dependency. A shared library would put something outside `rig/` on rig's path.
|
||||
**Check:** berth imports nothing from rig; `ports.sh` recomputes the port formula and agrees
|
||||
with rig's golden values.
|
||||
|
||||
**✖ B9 — "`langfuse.mcrn.ar` is an exception that cannot be generated."**
|
||||
*(`estate/mcrn.json`, `raw: true`)* Withdrawn 2026-09-14. It was filed as the one route a
|
||||
template could not express — a static `upstream{}` to a WireGuard address, with no `resolver`
|
||||
and no `set $var`. It was not an exception; it was **the first instance of the general case**.
|
||||
Those three properties are not three decisions, they are one: *this service is reached by
|
||||
address on the overlay, not by name on the docker network.* Naming that decision —
|
||||
`placement` — makes the file renderable.
|
||||
**Gone from:** `estate/mcrn.json` — `lng` and `langfuse` were **two entries for one socket**
|
||||
and are now one service with `placement: local`, `peer: nrft`, and a `local_host` for the
|
||||
name it answers to locally. `raw` is dropped.
|
||||
**Check:** the generated vhost matches the hand-written one, normalised for comments and
|
||||
whitespace — proof against a live route rather than an assertion.
|
||||
18
berth/ctrl/.env.example
Normal file
18
berth/ctrl/.env.example
Normal file
@@ -0,0 +1,18 @@
|
||||
# Machine-local config. Copy to ctrl/.env (gitignored) and edit.
|
||||
#
|
||||
# The estate's FACTS live in estate/<name>.json.
|
||||
# The provider's SHAPE lives in ctrl/env.d/<target>.env.
|
||||
# This file is only what differs between machines, plus credentials.
|
||||
|
||||
# ESTATE is required when estate/ holds more than one file.
|
||||
# ESTATE=mcrn
|
||||
# TARGET=aws
|
||||
|
||||
# ssh ALIASES, never hostnames — check.sh refuses a value containing a dot.
|
||||
# HOST=mcrn # app user, no sudo
|
||||
# HOST_ADMIN=mcrn-admin # sudo, only where genuinely required
|
||||
|
||||
# Credentials: names only, never values. The secrets stay in ~/.aws and
|
||||
# ~/.config/gcloud where their own tooling manages them.
|
||||
# AWS_PROFILE=default
|
||||
# GCP_PROJECT=
|
||||
58
berth/ctrl/certs.sh
Normal file
58
berth/ctrl/certs.sh
Normal file
@@ -0,0 +1,58 @@
|
||||
#!/usr/bin/env bash
|
||||
# The wildcard TLS cert for the gateway.
|
||||
#
|
||||
# Usage:
|
||||
# ./certs.sh status # SANs issued vs SANs the services need
|
||||
# ./certs.sh verify # inspect the cert served on :443
|
||||
# ./certs.sh renew # refuses: issues a real cert
|
||||
# ./certs.sh push # refuses: ships to a live gateway
|
||||
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")"
|
||||
|
||||
source ./lib/config.sh
|
||||
source ./lib/estate.sh
|
||||
load_config
|
||||
|
||||
status() {
|
||||
local issued; issued=$(estate_get "certs.issued" | python3 -c 'import json,sys
|
||||
try: print("\n".join(json.load(sys.stdin)))
|
||||
except Exception: pass')
|
||||
echo "issued SANs (estate/${ESTATE}.json: certs.issued):"
|
||||
echo "$issued" | sed 's/^/ /'
|
||||
echo
|
||||
echo "SANs the services NEED (derived from services[]):"
|
||||
estate_sans | sed 's/^/ /'
|
||||
echo
|
||||
local missing=0 s
|
||||
while IFS= read -r s; do
|
||||
[ -z "$s" ] && continue
|
||||
grep -qxF "$s" <<< "$issued" || { echo "MISSING: $s"; missing=1; }
|
||||
done < <(estate_sans)
|
||||
[ "$missing" = 0 ] && echo "the issued cert covers every derived name."
|
||||
echo
|
||||
echo "certbot image: $(eval echo "\$$CERTBOT_IMAGE_VAR") provider: $DNS_PROVIDER"
|
||||
}
|
||||
|
||||
verify() {
|
||||
echo "would run:"
|
||||
echo " echo | openssl s_client -connect ${DOMAIN}:443 -servername ${DOMAIN} 2>/dev/null \\"
|
||||
echo " | openssl x509 -noout -dates -ext subjectAltName"
|
||||
echo
|
||||
echo "read-only against a live host — announce and approve first (§7)."
|
||||
}
|
||||
|
||||
refuse() {
|
||||
echo "REFUSING: '$1' acts on a live cert and a live gateway." >&2
|
||||
echo " renew issues a real Let's Encrypt cert (rate-limited)." >&2
|
||||
echo " push rsyncs to the gateway and reloads nginx." >&2
|
||||
echo " Neither runs without explicit approval." >&2
|
||||
exit 1
|
||||
}
|
||||
|
||||
case "${1:-status}" in
|
||||
status) status ;;
|
||||
verify) verify ;;
|
||||
renew|push) refuse "$1" ;;
|
||||
*) echo "usage: $0 [status|verify|renew|push]" >&2; exit 1 ;;
|
||||
esac
|
||||
153
berth/ctrl/check.sh
Normal file
153
berth/ctrl/check.sh
Normal file
@@ -0,0 +1,153 @@
|
||||
#!/usr/bin/env bash
|
||||
# Is this estate coherent? Reports and instructs; never fixes.
|
||||
#
|
||||
# Takes no subcommand — there is one question to ask.
|
||||
# Every check corresponds to something wrong in the estate today.
|
||||
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")"
|
||||
|
||||
source ./lib/config.sh
|
||||
source ./lib/estate.sh
|
||||
load_config
|
||||
|
||||
WORST=0
|
||||
note() { echo " $*"; }
|
||||
warn() { echo " WARN $*"; [ "$WORST" -lt 1 ] && WORST=1; return 0; }
|
||||
bad() { echo " FAIL $*"; WORST=2; return 0; }
|
||||
|
||||
echo "estate: $ESTATE target: $TARGET domain: $DOMAIN"
|
||||
echo
|
||||
|
||||
# 1. cert coverage: a wildcard matches exactly one label.
|
||||
echo "certs — does the cert cover every name the services serve?"
|
||||
issued=$(estate_get "certs.issued" | python3 -c 'import json,sys
|
||||
try: print("\n".join(json.load(sys.stdin)))
|
||||
except Exception: pass')
|
||||
if [ -z "$issued" ]; then
|
||||
warn "no certs.issued in the estate file — cannot check coverage."
|
||||
else
|
||||
while IFS=$'\x1f' read -r name host up kind raw placement peer port lhost; do
|
||||
[ -z "$host" ] && continue
|
||||
fqdn="${host}.${DOMAIN}"
|
||||
# A '*' host stands for "any single label here" — check the deepest
|
||||
# name it can produce, which is the one that fails.
|
||||
probe="$fqdn"
|
||||
case "$host" in \*.*) probe="anyroom.${host#\*.}.${DOMAIN}" ;; esac
|
||||
covered=""
|
||||
while IFS= read -r san; do
|
||||
[ -z "$san" ] && continue
|
||||
if san_covers "$probe" "$san"; then covered=1; break; fi
|
||||
done <<< "$issued"
|
||||
if [ -z "$covered" ]; then
|
||||
bad "$name: '$probe' is covered by NO issued SAN"
|
||||
note " issued: $(echo "$issued" | tr '\n' ' ')"
|
||||
note " a wildcard matches exactly ONE label — reissue with"
|
||||
note " -d '*.${host#\*.}.${DOMAIN}' or move the name one level up"
|
||||
fi
|
||||
done < <(estate_services "$TARGET")
|
||||
[ "$WORST" -lt 2 ] && note "every service name is covered."
|
||||
fi
|
||||
echo
|
||||
|
||||
# 2. unmatched names: DNS and the cert are wildcard, nginx is exact, so
|
||||
# without a :443 default_server the fallback is whichever vhost loads first.
|
||||
echo "gateway — is there a deliberate answer for unmatched names?"
|
||||
DEFAULT_CONF="${PPL_DIR:-$HOME/wdir/semester/ppl}/gateway/nginx/conf.d/default.conf"
|
||||
if [ ! -f "$DEFAULT_CONF" ]; then
|
||||
note "ppl not on this machine at $DEFAULT_CONF — skipped."
|
||||
elif grep -qE '^\s*listen\s+443.*default_server' "$DEFAULT_CONF"; then
|
||||
note "default.conf has a :443 default_server."
|
||||
else
|
||||
bad "default.conf has NO :443 default_server."
|
||||
note " Unmatched names fall through to the first-loaded vhost."
|
||||
note " This is a PREREQUISITE for generating any config: adding a"
|
||||
note " generated include changes load order, and load order is what"
|
||||
note " currently decides the fallback."
|
||||
fi
|
||||
echo
|
||||
|
||||
# 3. a rule allowing a port nothing listens on is dead config; a service no
|
||||
# compose file declares is undocumented state. Needs both halves to see.
|
||||
echo "firewall — rules against listeners"
|
||||
estate_get "firewall" | python3 -c '
|
||||
import json,sys
|
||||
try: fw = json.load(sys.stdin)
|
||||
except Exception: fw = []
|
||||
for r in fw:
|
||||
n = r.get("note")
|
||||
print(" %-6s %-5s %s" % (r["port"], r.get("proto","tcp"), r.get("desc","")))
|
||||
if n: print(" UNRESOLVED: " + n)
|
||||
'
|
||||
note "listener side: unknown until captured (ss -ltnp over ssh $HOST)."
|
||||
# Ask the structure, not the prose: the condition is "does a peer still lack a
|
||||
# public key", not "is there a _status string". _status is ALWAYS non-empty —
|
||||
# capture rewrites it to "CAPTURED ..." — so testing it for emptiness pinned
|
||||
# this warning on permanently, including after the capture it asks for.
|
||||
uncaptured=$(estate_get "vpn.overlays.estate.peers" 2>/dev/null | python3 -c '
|
||||
import json, sys
|
||||
try:
|
||||
peers = json.load(sys.stdin)
|
||||
except Exception:
|
||||
sys.exit(0)
|
||||
print(" ".join(n for n, p in peers.items() if not p.get("public_key")))
|
||||
' 2>/dev/null)
|
||||
if [ -n "$uncaptured" ]; then
|
||||
warn "overlay: public keys not captured for:$uncaptured — see 'make vpn check'"
|
||||
note " 10.8.0.1 carries the registry and woodpecker gRPC;"
|
||||
note " 10.8.0.2 backs langfuse. Nothing in the tree creates the"
|
||||
note " interface — a fresh box cannot start the gateway compose."
|
||||
note " capture with: sudo wg show | make vpn capture --write"
|
||||
else
|
||||
note "overlay: $(estate_get 'vpn._status')"
|
||||
fi
|
||||
note "overlay detail: make vpn show estate"
|
||||
echo
|
||||
|
||||
# 4. ssh aliases, never hostnames: a bare hostname offers every agent key and
|
||||
# trips MaxAuthTries before reaching the right one.
|
||||
echo "ssh — aliases, never hostnames"
|
||||
for var in HOST HOST_ADMIN; do
|
||||
val="${!var:-}"
|
||||
if [ -z "$val" ]; then
|
||||
warn "$var is unset."
|
||||
elif [[ "$val" == *.* ]]; then
|
||||
bad "$var='$val' looks like a hostname, not a ~/.ssh/config alias."
|
||||
note " A bare hostname trips MaxAuthTries before reaching the key."
|
||||
elif [ -f "$HOME/.ssh/config" ] && grep -qiE "^\s*Host\s+.*\b${val}\b" "$HOME/.ssh/config"; then
|
||||
note "$var=$val — Host block present."
|
||||
else
|
||||
warn "$var='$val' has no matching Host block in ~/.ssh/config."
|
||||
fi
|
||||
done
|
||||
for f in "$HOME/wdir/semester/ppl/ctrl/.env"; do
|
||||
[ -f "$f" ] || continue
|
||||
if grep -qE '^SERVER=.*\.' "$f"; then
|
||||
warn "$f sets SERVER to a hostname, not an alias — every ppl script inherits it."
|
||||
fi
|
||||
done
|
||||
echo
|
||||
|
||||
# ── 5. toolchain ───────────────────────────────────────────────────────────
|
||||
echo "toolchain"
|
||||
for t in python3 "$TOFU_BIN" aws gcloud ssh rsync wg; do
|
||||
if command -v "$t" >/dev/null 2>&1; then
|
||||
note "$(printf '%-8s' "$t") present"
|
||||
else
|
||||
note "$(printf '%-8s' "$t") MISSING — $(case $t in
|
||||
tofu) echo 'blocks: estate plan/apply; terraform works identically' ;;
|
||||
wg) echo 'blocks: vpn keygen and the overlay checks' ;;
|
||||
aws) echo 'blocks: the control-plane inventory, dns on route53' ;;
|
||||
gcloud) echo 'blocks: the gcp estate' ;;
|
||||
*) echo 'blocks: most things' ;;
|
||||
esac)"
|
||||
fi
|
||||
done
|
||||
echo
|
||||
|
||||
case "$WORST" in
|
||||
0) echo "OK" ;;
|
||||
1) echo "OK, with warnings" ;;
|
||||
2) echo "PROBLEMS FOUND — see FAIL lines above" ;;
|
||||
esac
|
||||
exit 0
|
||||
81
berth/ctrl/dns.sh
Normal file
81
berth/ctrl/dns.sh
Normal file
@@ -0,0 +1,81 @@
|
||||
#!/usr/bin/env bash
|
||||
# DNS records, over whichever provider the target names.
|
||||
#
|
||||
# Usage:
|
||||
# ./dns.sh list
|
||||
# ./dns.sh add <subdomain> # <sub>.<domain> -> <domain>
|
||||
# ./dns.sh add-wildcard <sub>
|
||||
# ./dns.sh remove <subdomain>
|
||||
#
|
||||
# Read-only verbs announce and wait; mutating ones refuse.
|
||||
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")"
|
||||
|
||||
source ./lib/config.sh
|
||||
source ./lib/estate.sh
|
||||
load_config
|
||||
|
||||
announce() {
|
||||
echo "would run:"
|
||||
printf ' %s\n' "$*"
|
||||
}
|
||||
|
||||
list() {
|
||||
case "$DNS_PROVIDER" in
|
||||
route53)
|
||||
announce "aws route53 list-resource-record-sets --hosted-zone-id $AWS_HOSTED_ZONE_ID" \
|
||||
"--query 'ResourceRecordSets[].[Name,Type,TTL,ResourceRecords[0].Value]' --output table"
|
||||
;;
|
||||
google)
|
||||
announce "gcloud dns record-sets list --zone=${ESTATE}-zone --project=${GCP_PROJECT:-<unset>}"
|
||||
;;
|
||||
*) echo "unknown DNS_PROVIDER: $DNS_PROVIDER" >&2; exit 1 ;;
|
||||
esac
|
||||
echo
|
||||
echo "read-only, but NOT run: each batch is announced and"
|
||||
echo "waited on. Approve it and it runs."
|
||||
}
|
||||
|
||||
mutate() {
|
||||
local verb="$1" sub="${2:-}"
|
||||
[ -z "$sub" ] && { echo "usage: $0 $verb <subdomain>" >&2; exit 1; }
|
||||
local name
|
||||
case "$verb" in
|
||||
add) name="${sub}.${DOMAIN}" ;;
|
||||
add-wildcard) name="*.${sub}.${DOMAIN}" ;;
|
||||
remove) name="${sub}.${DOMAIN}" ;;
|
||||
esac
|
||||
echo "$verb: $name -> $DOMAIN (provider: $DNS_PROVIDER)"
|
||||
echo
|
||||
|
||||
# The wildcard-depth rule again, applied BEFORE the record is created rather
|
||||
# than discovered in a browser afterwards.
|
||||
if [ "$verb" = "add" ]; then
|
||||
local issued; issued=$(estate_get "certs.issued" | python3 -c 'import json,sys
|
||||
try: print("\n".join(json.load(sys.stdin)))
|
||||
except Exception: pass')
|
||||
local covered=""
|
||||
while IFS= read -r san; do
|
||||
[ -z "$san" ] && continue
|
||||
san_covers "$name" "$san" && { covered=1; break; }
|
||||
done <<< "$issued"
|
||||
[ -z "$covered" ] && {
|
||||
echo "WARNING: '$name' is covered by no issued SAN." >&2
|
||||
echo " A wildcard matches ONE label. The record would" >&2
|
||||
echo " resolve and then fail TLS. Reissue the cert first." >&2
|
||||
echo >&2
|
||||
}
|
||||
fi
|
||||
|
||||
echo "REFUSING: this changes live DNS." >&2
|
||||
echo " Nothing is created, modified or deleted on any account without" >&2
|
||||
echo " explicit approval for that specific action." >&2
|
||||
exit 1
|
||||
}
|
||||
|
||||
case "${1:-list}" in
|
||||
list) list ;;
|
||||
add|add-wildcard|remove) mutate "$@" ;;
|
||||
*) echo "usage: $0 [list|add <sub>|add-wildcard <sub>|remove <sub>]" >&2; exit 1 ;;
|
||||
esac
|
||||
27
berth/ctrl/docs.sh
Normal file
27
berth/ctrl/docs.sh
Normal file
@@ -0,0 +1,27 @@
|
||||
#!/usr/bin/env bash
|
||||
# Documentation.
|
||||
#
|
||||
# Usage:
|
||||
# ./docs.sh serve|graphs
|
||||
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")"
|
||||
|
||||
source ./lib/config.sh
|
||||
source ./lib/estate.sh
|
||||
load_config
|
||||
|
||||
case "${1:-serve}" in
|
||||
serve)
|
||||
echo "berth has no doc server yet — the README is the documentation."
|
||||
echo " $(cd .. && pwd)/README.md"
|
||||
echo
|
||||
echo "the estate, resolved: make estate show"
|
||||
;;
|
||||
graphs)
|
||||
echo "not implemented. The estate file is the graph's source; a"
|
||||
echo "renderer belongs with docgen/graphgen, which is another thread's"
|
||||
echo "another thread's — so this is a handoff, not a stub to fill in here."
|
||||
;;
|
||||
*) echo "usage: $0 [serve|graphs]" >&2; exit 1 ;;
|
||||
esac
|
||||
14
berth/ctrl/env.d/aws.env
Normal file
14
berth/ctrl/env.d/aws.env
Normal file
@@ -0,0 +1,14 @@
|
||||
# Target: AWS — the estate that actually runs today (mcrn.ar).
|
||||
TARGET_NAME=aws
|
||||
CLOUD=aws
|
||||
REGION=us-east-1
|
||||
INSTANCE_TYPE=t3.small
|
||||
|
||||
# The Route53 hosted zone. It ALREADY EXISTS and is reused, never created —
|
||||
# see estate.sh's refusal to create a zone on this target.
|
||||
AWS_HOSTED_ZONE_ID=Z02279903503ZMIB5FC1N
|
||||
SSH_KEY_NAME=mcrn
|
||||
|
||||
# certbot's DNS-01 plugin for this provider.
|
||||
CERTBOT_IMAGE_VAR=CERTBOT_AWS_IMAGE
|
||||
DNS_PROVIDER=route53
|
||||
17
berth/ctrl/env.d/gcp.env
Normal file
17
berth/ctrl/env.d/gcp.env
Normal file
@@ -0,0 +1,17 @@
|
||||
# Target: GCP — the replica, on its own domain.
|
||||
#
|
||||
# The domain differs from AWS's on purpose: this target CREATES a DNS zone
|
||||
# where aws reuses one, so a shared domain would create a second authoritative
|
||||
# zone and break live DNS. The domain lives in estate/nrft.json, not here.
|
||||
TARGET_NAME=gcp
|
||||
CLOUD=gcp
|
||||
REGION=us-central1
|
||||
INSTANCE_TYPE=e2-small
|
||||
|
||||
GCP_PROJECT=
|
||||
GCP_ZONE=us-central1-a
|
||||
|
||||
# Delegation order: create the zone and let it answer first, then set the
|
||||
# nameservers at the registrar — it validates that they respond.
|
||||
CERTBOT_IMAGE_VAR=CERTBOT_GCP_IMAGE
|
||||
DNS_PROVIDER=google
|
||||
101
berth/ctrl/estate.sh
Normal file
101
berth/ctrl/estate.sh
Normal file
@@ -0,0 +1,101 @@
|
||||
#!/usr/bin/env bash
|
||||
# The estate: what it is, what would change, and — behind a gate — changing it.
|
||||
#
|
||||
# Usage:
|
||||
# ./estate.sh # show
|
||||
# ./estate.sh list # every estate/*.json
|
||||
# ./estate.sh plan # tofu plan, read-only
|
||||
# ./estate.sh apply --yes # refuses without --yes
|
||||
# ./estate.sh destroy --yes
|
||||
#
|
||||
# Why the default is read-only: ../README.md
|
||||
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")"
|
||||
|
||||
source ./lib/config.sh
|
||||
source ./lib/estate.sh
|
||||
load_config
|
||||
|
||||
show() {
|
||||
echo "estate: $ESTATE ($ESTATE_FILE)"
|
||||
echo "target: $TARGET (cloud=$CLOUD region=$REGION)"
|
||||
echo "domain: $DOMAIN"
|
||||
echo "host: $HOST (sudo: $HOST_ADMIN)"
|
||||
echo "workspace: $TOFU_WORKSPACE"
|
||||
echo
|
||||
local status; status=$(estate_get "_meta.status")
|
||||
[ -n "$status" ] && echo " !! $status" && echo
|
||||
|
||||
echo "services in scope for '$TARGET':"
|
||||
local name host up kind raw
|
||||
while IFS=$'\x1f' read -r name host up kind raw placement peer port lhost; do
|
||||
[ -z "$name" ] && continue
|
||||
printf ' %-12s %-14s %-24s %s%s\n' \
|
||||
"$name" "${host:--}" "${up:--}" "$kind" \
|
||||
"$([ -n "$raw" ] && echo ' [hand-written]')"
|
||||
done < <(estate_services "$TARGET")
|
||||
|
||||
echo
|
||||
echo "cert SANs (derived, not listed):"
|
||||
estate_sans | sed 's/^/ /'
|
||||
}
|
||||
|
||||
list() {
|
||||
local f n
|
||||
printf '%-12s %-16s %s\n' ESTATE DOMAIN STATUS
|
||||
for f in ../estate/*.json; do
|
||||
[ -f "$f" ] || continue
|
||||
n=$(basename "$f" .json)
|
||||
printf '%-12s %-16s %s%s\n' "$n" \
|
||||
"$(python3 -c 'import json,sys;print(json.load(open(sys.argv[1])).get("domain",""))' "$f")" \
|
||||
"$(python3 -c 'import json,sys;print(json.load(open(sys.argv[1])).get("_meta",{}).get("status",""))' "$f")" \
|
||||
"$([ "$n" = "$ESTATE" ] && echo ' <- this one')"
|
||||
done
|
||||
}
|
||||
|
||||
# Read-only. Meaningful only once state is imported: against empty state,
|
||||
# plan reports "create N resources", which is not drift.
|
||||
plan() {
|
||||
echo "== $TOFU_BIN plan =="
|
||||
if ! command -v "$TOFU_BIN" >/dev/null; then
|
||||
echo " $TOFU_BIN not installed — skipped." >&2
|
||||
echo " OpenTofu is the MPL-2.0 fork; 'terraform' works identically." >&2
|
||||
else
|
||||
echo " would run: $TOFU_BIN plan -var-file=<(estate)"
|
||||
fi
|
||||
echo
|
||||
echo "NOTE: not wired up yet, and that is the point. tofu plan against empty"
|
||||
echo " state reports \"create N resources\" — which is not drift, it is an"
|
||||
echo " empty state. It becomes the check that proves the description"
|
||||
echo " matches reality only once state is IMPORTED from the inventory."
|
||||
echo " ppl/infra/ describes an aspiration: it was never applied."
|
||||
}
|
||||
|
||||
# The gate. Two things have to be true: --yes present, AND the plan shown first.
|
||||
require_yes() {
|
||||
local verb="$1"; shift
|
||||
local yes=""
|
||||
for a in "$@"; do [ "$a" = "--yes" ] && yes=1; done
|
||||
if [ -z "$yes" ]; then
|
||||
echo "refusing to $verb without --yes." >&2
|
||||
echo >&2
|
||||
echo " $verb changes a live, billable estate and can take DNS with it." >&2
|
||||
echo " Read the plan first: make estate plan" >&2
|
||||
echo " Then: ./ctrl/estate.sh $verb --yes" >&2
|
||||
exit 1
|
||||
fi
|
||||
echo "refusing to $verb: not implemented, and deliberately so." >&2
|
||||
echo " The executor is not wired up, and nothing is imported yet, so" >&2
|
||||
echo " there is nothing truthful to apply." >&2
|
||||
exit 1
|
||||
}
|
||||
|
||||
case "${1:-show}" in
|
||||
show) show ;;
|
||||
list) list ;;
|
||||
plan) plan ;;
|
||||
apply) shift; require_yes apply "$@" ;;
|
||||
destroy) shift; require_yes destroy "$@" ;;
|
||||
*) echo "usage: $0 [show|list|plan|apply --yes|destroy --yes]" >&2; exit 1 ;;
|
||||
esac
|
||||
63
berth/ctrl/host.sh
Normal file
63
berth/ctrl/host.sh
Normal file
@@ -0,0 +1,63 @@
|
||||
#!/usr/bin/env bash
|
||||
# The remote box — the other half of the inventory.
|
||||
#
|
||||
# Usage:
|
||||
# ./host.sh status|ports|services
|
||||
#
|
||||
# Announces what it would run over `ssh $HOST` and does not run it. Always the
|
||||
# ssh alias, never a hostname: there is no `Host mcrn.ar` block, so a bare
|
||||
# hostname offers every agent key and trips MaxAuthTries.
|
||||
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")"
|
||||
|
||||
source ./lib/config.sh
|
||||
source ./lib/estate.sh
|
||||
load_config
|
||||
|
||||
guard_alias() {
|
||||
if [[ "$HOST" == *.* ]]; then
|
||||
echo "HOST='$HOST' is a hostname, not a ~/.ssh/config alias. Refusing." >&2
|
||||
exit 1
|
||||
fi
|
||||
}
|
||||
|
||||
announce_batch() {
|
||||
echo "would run, over 'ssh $HOST':"
|
||||
printf ' %s\n' "$@"
|
||||
echo
|
||||
echo "read-only, and NOT run. Announce-first applies to EACH batch, not"
|
||||
echo "once per session — so the commands can be read and"
|
||||
echo "learned rather than scrolled past."
|
||||
}
|
||||
|
||||
guard_alias
|
||||
case "${1:-status}" in
|
||||
status)
|
||||
announce_batch \
|
||||
"uname -a; uptime; df -h /" \
|
||||
"docker ps --format '{{.Names}}\t{{.Image}}\t{{.Ports}}'" \
|
||||
"docker network inspect gateway --format '{{range .Containers}}{{.Name}} {{end}}'" \
|
||||
"systemctl list-units --type=service --state=running --no-pager" \
|
||||
"systemctl list-timers --no-pager"
|
||||
echo
|
||||
echo "sudo-only, over 'ssh $HOST_ADMIN' and only where genuinely needed:"
|
||||
echo " wg show # the WireGuard peers that exist nowhere in the tree"
|
||||
;;
|
||||
ports)
|
||||
announce_batch "ss -ltnp"
|
||||
echo "the estate declares these firewall rules:"
|
||||
estate_get "firewall" | python3 -c '
|
||||
import json,sys
|
||||
for r in json.load(sys.stdin):
|
||||
print(" %-6s %-5s %s" % (r["port"], r.get("proto","tcp"), r.get("desc","")))'
|
||||
;;
|
||||
services)
|
||||
announce_batch "docker compose -f ~/ppl/gateway/docker-compose.yml ps"
|
||||
echo "the estate declares $(estate_services "$TARGET" | grep -c . ) service(s) for target '$TARGET'."
|
||||
echo "the gateway compose declares 8. The difference is sibling repos'"
|
||||
echo "stacks joining the shared 'gateway' network — intended design, but"
|
||||
echo "nothing in the tree lists it. The inventory produces that list."
|
||||
;;
|
||||
*) echo "usage: $0 [status|ports|services]" >&2; exit 1 ;;
|
||||
esac
|
||||
94
berth/ctrl/lib/config.sh
Normal file
94
berth/ctrl/lib/config.sh
Normal file
@@ -0,0 +1,94 @@
|
||||
# Shared config loading. Sourced, never executed. Run from ctrl/.
|
||||
#
|
||||
# Precedence, weakest first:
|
||||
# ctrl/versions.env pinned toolchain (committed)
|
||||
# ctrl/env.d/<target>.env provider shape: aws|gcp (committed)
|
||||
# ctrl/.env machine-local + secrets (gitignored)
|
||||
# the caller's env `make estate plan TARGET=gcp` (always wins)
|
||||
|
||||
CONFIG_OVERRIDABLE="TARGET ESTATE CLOUD REGION INSTANCE_TYPE
|
||||
HOST HOST_ADMIN AWS_PROFILE AWS_HOSTED_ZONE_ID
|
||||
GCP_PROJECT GCP_ZONE TOFU_WORKSPACE"
|
||||
|
||||
# rig's port formula, reproduced rather than imported. cksum because it is
|
||||
# POSIX and gives the same value on every machine. Used to verify, not allocate.
|
||||
derive_port_base() {
|
||||
local h; h=$(printf '%s' "$1" | cksum | awk '{print $1}')
|
||||
echo $((20000 + (h % 200) * 10))
|
||||
}
|
||||
|
||||
_config_restore() {
|
||||
local line
|
||||
while IFS= read -r line; do
|
||||
if [ -n "$line" ]; then
|
||||
eval "export $line"
|
||||
fi
|
||||
done <<< "$1"
|
||||
# A while loop returns its last body command's status; the trailing empty
|
||||
# line would otherwise make this return 1 and trip `set -e` in the caller.
|
||||
return 0
|
||||
}
|
||||
|
||||
load_config() {
|
||||
local k saved=""
|
||||
for k in $CONFIG_OVERRIDABLE; do
|
||||
# ${!k+x} distinguishes "set but empty" from "unset" — an explicit
|
||||
# FOO= on the command line is a real choice and must survive.
|
||||
if [ -n "${!k+x}" ]; then
|
||||
saved+="$k=$(printf '%q' "${!k}")"$'\n'
|
||||
fi
|
||||
done
|
||||
|
||||
set -a
|
||||
source ./versions.env
|
||||
[ -f ./.env ] && source ./.env
|
||||
set +a
|
||||
|
||||
# Re-apply overrides now so TARGET is the caller's before we pick the file.
|
||||
_config_restore "$saved"
|
||||
|
||||
local target="${TARGET:-aws}"
|
||||
if [ ! -f "./env.d/${target}.env" ]; then
|
||||
echo "no such target: env.d/${target}.env" >&2
|
||||
echo "available: $(ls env.d/*.env 2>/dev/null | xargs -n1 basename | sed 's/\.env$//' | tr '\n' ' ')" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
set -a
|
||||
source "./env.d/${target}.env"
|
||||
[ -f ./.env ] && source ./.env
|
||||
set +a
|
||||
|
||||
_config_restore "$saved"
|
||||
|
||||
TARGET="$target"
|
||||
|
||||
# Identity is explicit: berth never guesses which estate it is acting on.
|
||||
# The one convenience: a single estate/*.json is used without being asked.
|
||||
if [ -z "${ESTATE:-}" ]; then
|
||||
local n; n=$(ls ../estate/*.json 2>/dev/null | wc -l)
|
||||
if [ "$n" = "1" ]; then
|
||||
ESTATE=$(basename "$(ls ../estate/*.json)" .json)
|
||||
else
|
||||
echo "ESTATE is not set and estate/ holds $n candidates — refusing to guess." >&2
|
||||
echo "available: $(ls ../estate/*.json 2>/dev/null | xargs -n1 basename | sed 's/\.json$//' | tr '\n' ' ')" >&2
|
||||
echo "set it: make estate show ESTATE=<name>, or ESTATE= in ctrl/.env" >&2
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
|
||||
ESTATE_FILE="../estate/${ESTATE}.json"
|
||||
if [ ! -f "$ESTATE_FILE" ]; then
|
||||
echo "no such estate: estate/${ESTATE}.json" >&2
|
||||
echo "available: $(ls ../estate/*.json 2>/dev/null | xargs -n1 basename | sed 's/\.json$//' | tr '\n' ' ')" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Facts come from the estate file, never restated in a target env.
|
||||
DOMAIN=$(estate_get "domain")
|
||||
HOST="${HOST:-$(estate_get "host")}"
|
||||
HOST_ADMIN="${HOST_ADMIN:-$(estate_get "host_admin")}"
|
||||
|
||||
# Workspace == target, so the two can never mean different things.
|
||||
TOFU_WORKSPACE="${TOFU_WORKSPACE:-$TARGET}"
|
||||
}
|
||||
147
berth/ctrl/lib/estate.sh
Normal file
147
berth/ctrl/lib/estate.sh
Normal file
@@ -0,0 +1,147 @@
|
||||
# Reading and projecting estate/<name>.json. Sourced, never executed.
|
||||
#
|
||||
# python3 rather than jq: berth's floor already includes python3, so it is a
|
||||
# dependency berth has rather than one it adds.
|
||||
|
||||
estate_get() {
|
||||
python3 -c '
|
||||
import json, sys
|
||||
d = json.load(open(sys.argv[1]))
|
||||
for k in sys.argv[2].split("."):
|
||||
if isinstance(d, list):
|
||||
try: k = int(k)
|
||||
except ValueError: sys.exit(0)
|
||||
try: d = d[k]
|
||||
except Exception: sys.exit(0)
|
||||
print("" if d is None else d if isinstance(d, str) else json.dumps(d))
|
||||
' "$ESTATE_FILE" "$1"
|
||||
}
|
||||
|
||||
# Services in scope for one target. A service names its targets; absent = all.
|
||||
# Fields are US-separated (0x1f), not tab: tab is IFS whitespace, so bash
|
||||
# collapses a run of them and an empty field would shift every later column.
|
||||
estate_services() {
|
||||
local target="${1:-$TARGET}"
|
||||
python3 -c '
|
||||
import json, sys
|
||||
d = json.load(open(sys.argv[1]))
|
||||
target = sys.argv[2]
|
||||
for s in d.get("services", []):
|
||||
tg = s.get("targets")
|
||||
if tg is not None and target not in tg:
|
||||
continue
|
||||
print("\x1f".join([
|
||||
s.get("name", ""),
|
||||
s.get("host", ""),
|
||||
str(s.get(target + "_upstream", s.get("upstream", "")) or ""),
|
||||
s.get("kind", "proxy"),
|
||||
"raw" if s.get("raw") else "",
|
||||
s.get("placement", "box"),
|
||||
s.get("peer", ""),
|
||||
str(s.get("port", "") or ""),
|
||||
s.get("local_host", s.get("host", "")),
|
||||
]))
|
||||
' "$ESTATE_FILE" "$target"
|
||||
}
|
||||
|
||||
# The SAN list the services need, derived — never a literal list.
|
||||
estate_sans() {
|
||||
python3 -c '
|
||||
import json, sys
|
||||
d = json.load(open(sys.argv[1]))
|
||||
domain = d["domain"]
|
||||
sans = [domain]
|
||||
depths = set()
|
||||
for s in d.get("services", []):
|
||||
h = s.get("host", "")
|
||||
if not h:
|
||||
continue
|
||||
# A wildcard matches exactly ONE label. "git" needs *.domain; "dlt.spr"
|
||||
# needs *.spr.domain. The parent of the leaf is what has to be covered.
|
||||
parent = h.split(".", 1)[1] if "." in h else ""
|
||||
depths.add(parent)
|
||||
for p in sorted(depths):
|
||||
sans.append("*." + (p + "." if p else "") + domain)
|
||||
for s in sans:
|
||||
print(s)
|
||||
' "$ESTATE_FILE"
|
||||
}
|
||||
|
||||
# Is <fqdn> covered by <san>? A wildcard matches exactly one label.
|
||||
san_covers() {
|
||||
local fqdn="$1" san="$2"
|
||||
[ "$fqdn" = "$san" ] && return 0
|
||||
case "$san" in
|
||||
\*.*)
|
||||
local suffix="${san#\*.}"
|
||||
# Must end in .suffix AND have exactly one extra label.
|
||||
case "$fqdn" in
|
||||
*".$suffix") [ "${fqdn%".$suffix"}" = "${fqdn%%.*}" ] && return 0 ;;
|
||||
esac
|
||||
;;
|
||||
esac
|
||||
return 1
|
||||
}
|
||||
|
||||
# ── the overlay ────────────────────────────────────────────────────────────
|
||||
|
||||
# Every overlay name, one per line.
|
||||
overlay_names() {
|
||||
python3 -c '
|
||||
import json, sys
|
||||
d = json.load(open(sys.argv[1]))
|
||||
for n in d.get("vpn", {}).get("overlays", {}):
|
||||
print(n)
|
||||
' "$ESTATE_FILE"
|
||||
}
|
||||
|
||||
# Peers of one overlay, US-separated:
|
||||
# name, address, role, endpoint, public_key, allowed_ips, keepalive
|
||||
overlay_peers() {
|
||||
python3 -c '
|
||||
import json, sys
|
||||
d = json.load(open(sys.argv[1]))
|
||||
ov = d.get("vpn", {}).get("overlays", {}).get(sys.argv[2], {})
|
||||
for name, p in ov.get("peers", {}).items():
|
||||
print("\x1f".join(str(x) if x is not None else "" for x in [
|
||||
name, p.get("address"), p.get("role"), p.get("endpoint"),
|
||||
p.get("public_key"), p.get("allowed_ips"), p.get("keepalive"),
|
||||
]))
|
||||
' "$ESTATE_FILE" "$1"
|
||||
}
|
||||
|
||||
overlay_get() { estate_get "vpn.overlays.$1.$2"; }
|
||||
|
||||
# Is an address inside a CIDR? Pure python so there is no ipcalc dependency —
|
||||
# berth's floor already includes python3 because the IaC side needs it.
|
||||
addr_in_subnet() {
|
||||
python3 -c '
|
||||
import ipaddress, sys
|
||||
try:
|
||||
sys.exit(0 if ipaddress.ip_address(sys.argv[1]) in ipaddress.ip_network(sys.argv[2], strict=False) else 1)
|
||||
except ValueError:
|
||||
sys.exit(2)
|
||||
' "$1" "$2"
|
||||
}
|
||||
|
||||
# The upstream a service actually resolves to, as "host:port".
|
||||
#
|
||||
# A PLACED service has no literal `upstream` field: ✖ B9 replaced langfuse's
|
||||
# hand-written `10.8.0.2:3000` with placement+peer+port, because being reached
|
||||
# by address on the overlay is ONE decision, not three properties. Everything
|
||||
# that asks "what does this service point at" must therefore resolve it the
|
||||
# same way, or it silently sees an empty string and skips the service — which
|
||||
# is exactly how vpn.sh's bindings invariant went quiet after B9 landed.
|
||||
#
|
||||
# usage: service_upstream <up> <placement> <peer> <port>
|
||||
service_upstream() {
|
||||
local up="$1" placement="$2" peer="$3" port="$4"
|
||||
case "$placement" in
|
||||
local|instance)
|
||||
local addr; addr="$(overlay_get estate "peers.${peer}.address")"
|
||||
[ -z "$addr" ] && return 1
|
||||
printf '%s:%s' "$addr" "$port"
|
||||
;;
|
||||
*) printf '%s' "$up" ;;
|
||||
esac
|
||||
}
|
||||
78
berth/ctrl/ports.sh
Normal file
78
berth/ctrl/ports.sh
Normal file
@@ -0,0 +1,78 @@
|
||||
#!/usr/bin/env bash
|
||||
# The local port map, and whether it still agrees with rig.
|
||||
#
|
||||
# Usage:
|
||||
# ./ports.sh show # DERIVED / ACTIVE / SOURCE
|
||||
# ./ports.sh verify # recompute rig's formula, report drift
|
||||
#
|
||||
# berth recomputes rig's port formula rather than importing it, so neither
|
||||
# depends on the other. See ../README.md.
|
||||
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")"
|
||||
|
||||
source ./lib/config.sh
|
||||
source ./lib/estate.sh
|
||||
load_config
|
||||
|
||||
# name<US>host<US>local_port for everything that has one.
|
||||
_local_ports() {
|
||||
python3 -c '
|
||||
import json, sys
|
||||
d = json.load(open(sys.argv[1]))
|
||||
for s in d.get("services", []):
|
||||
p = s.get("local_port")
|
||||
if p:
|
||||
print("\x1f".join([s.get("name",""), s.get("host",""), str(p)]))
|
||||
' "$ESTATE_FILE"
|
||||
}
|
||||
|
||||
show() {
|
||||
local ld; ld=$(estate_get "local_domain"); : "${ld:=local.ar}"
|
||||
printf '%-12s %-22s %-8s %-8s %s\n' NAME ADDRESS ACTIVE DERIVED SOURCE
|
||||
local name host port base
|
||||
while IFS=$'\x1f' read -r name host port; do
|
||||
[ -z "$name" ] && continue
|
||||
base=$(derive_port_base "$host")
|
||||
if [ "$port" = "$base" ]; then
|
||||
printf '%-12s %-22s %-8s %-8s %s\n' "$name" "${host}.${ld}" "$port" "$base" "derived"
|
||||
else
|
||||
printf '%-12s %-22s %-8s %-8s %s\n' "$name" "${host}.${ld}" "$port" "$base" "override"
|
||||
fi
|
||||
done < <(_local_ports)
|
||||
echo
|
||||
echo "DERIVED is what rig's formula gives for that name. ACTIVE is what the"
|
||||
echo "estate records. 'override' is not an error — most of these were never"
|
||||
echo "rigs. 'make ports verify' says which ones should have matched."
|
||||
}
|
||||
|
||||
verify() {
|
||||
local name host port base rc=0 checked=0
|
||||
while IFS=$'\x1f' read -r name host port; do
|
||||
[ -z "$name" ] && continue
|
||||
# Only 20000-21999 is rig's to predict; anything else was never derived.
|
||||
if [ "$port" -lt 20000 ] || [ "$port" -gt 21999 ]; then
|
||||
continue
|
||||
fi
|
||||
checked=$((checked + 1))
|
||||
base=$(derive_port_base "$host")
|
||||
if [ "$port" != "$base" ]; then
|
||||
echo "DRIFT $name (${host}): estate says $port, rig's formula gives $base"
|
||||
echo " either the rig pinned HTTP_PORT in its ctrl/.env, or the"
|
||||
echo " folder was renamed. The Caddy map is stale either way."
|
||||
rc=1
|
||||
else
|
||||
echo "ok $name (${host}): $port"
|
||||
fi
|
||||
done < <(_local_ports)
|
||||
echo
|
||||
echo "checked $checked rig-shaped port(s) in 20000-21999."
|
||||
[ "$rc" = 0 ] && echo "no drift." || echo "drift found — regenerate with 'make services render local'."
|
||||
return 0
|
||||
}
|
||||
|
||||
case "${1:-show}" in
|
||||
show) show ;;
|
||||
verify) verify ;;
|
||||
*) echo "usage: $0 [show|verify]" >&2; exit 1 ;;
|
||||
esac
|
||||
33
berth/ctrl/registry.sh
Normal file
33
berth/ctrl/registry.sh
Normal file
@@ -0,0 +1,33 @@
|
||||
#!/usr/bin/env bash
|
||||
# The image registry — remote, and reachable only over the overlay.
|
||||
#
|
||||
# Usage:
|
||||
# ./registry.sh status
|
||||
#
|
||||
# One verb: berth reports on a registry running on someone else's box. Starting
|
||||
# and stopping it is that box's business.
|
||||
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")"
|
||||
|
||||
source ./lib/config.sh
|
||||
source ./lib/estate.sh
|
||||
load_config
|
||||
|
||||
status() {
|
||||
local wg_server; wg_server=$(overlay_get estate "peers.box.address")
|
||||
echo "registry: registry.${DOMAIN} (public pull via /v2/)"
|
||||
echo "push: ${wg_server}:5000 (WireGuard-only, not in the firewall)"
|
||||
echo
|
||||
echo "would run:"
|
||||
echo " curl -s https://registry.${DOMAIN}/v2/_catalog"
|
||||
echo
|
||||
echo "NOTE: the push endpoint binds ${wg_server}, and $(estate_get 'vpn._status')."
|
||||
echo " A freshly-provisioned box cannot start the gateway compose file"
|
||||
echo " at all, because that bind fails."
|
||||
}
|
||||
|
||||
case "${1:-status}" in
|
||||
status) status ;;
|
||||
*) echo "usage: $0 [status]" >&2; exit 1 ;;
|
||||
esac
|
||||
32
berth/ctrl/render/nginx-proxy.tmpl
Normal file
32
berth/ctrl/render/nginx-proxy.tmpl
Normal file
@@ -0,0 +1,32 @@
|
||||
# ${NAME} — GENERATED by berth from estate/${ESTATE}.json. Do not edit.
|
||||
# Edit the estate file and re-run: make services render aws
|
||||
|
||||
server {
|
||||
listen 80;
|
||||
server_name ${FQDN};
|
||||
return 301 https://$host$request_uri;
|
||||
}
|
||||
|
||||
server {
|
||||
listen 443 ssl;
|
||||
server_name ${FQDN};
|
||||
|
||||
ssl_certificate /etc/nginx/certs/live/${DOMAIN}/fullchain.pem;
|
||||
ssl_certificate_key /etc/nginx/certs/live/${DOMAIN}/privkey.pem;
|
||||
|
||||
# Docker's embedded DNS. Naming the upstream in a VARIABLE forces runtime
|
||||
# resolution, so nginx STARTS even when the upstream container is absent.
|
||||
# With a literal proxy_pass, one stopped container takes the whole gateway
|
||||
# down at reload — which is what makes one nginx able to front a dozen
|
||||
# independent compose stacks.
|
||||
resolver 127.0.0.11 valid=30s;
|
||||
|
||||
location / {
|
||||
set $upstream_${NAME} ${UPSTREAM_HOST};
|
||||
proxy_pass http://$upstream_${NAME}:${UPSTREAM_PORT};
|
||||
proxy_set_header Host $host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $scheme;
|
||||
}
|
||||
}
|
||||
23
berth/ctrl/render/nginx-static.tmpl
Normal file
23
berth/ctrl/render/nginx-static.tmpl
Normal file
@@ -0,0 +1,23 @@
|
||||
# ${NAME} — GENERATED by berth from estate/${ESTATE}.json. Do not edit.
|
||||
# Edit the estate file and re-run: make services render aws
|
||||
|
||||
server {
|
||||
listen 80;
|
||||
server_name ${FQDN};
|
||||
return 301 https://$host$request_uri;
|
||||
}
|
||||
|
||||
server {
|
||||
listen 443 ssl;
|
||||
server_name ${FQDN};
|
||||
|
||||
ssl_certificate /etc/nginx/certs/live/${DOMAIN}/fullchain.pem;
|
||||
ssl_certificate_key /etc/nginx/certs/live/${DOMAIN}/privkey.pem;
|
||||
|
||||
root /usr/share/nginx/html/${NAME};
|
||||
index index.html;
|
||||
|
||||
location / {
|
||||
try_files $uri $uri/ =404;
|
||||
}
|
||||
}
|
||||
32
berth/ctrl/render/nginx-upstream.tmpl
Normal file
32
berth/ctrl/render/nginx-upstream.tmpl
Normal file
@@ -0,0 +1,32 @@
|
||||
# ${NAME} — GENERATED by berth from estate/${ESTATE}.json. Do not edit.
|
||||
# Placement: ${PLACEMENT} (${PEER}) — reached over the overlay, not the docker network.
|
||||
|
||||
upstream ${NAME}_backend {
|
||||
server ${UPSTREAM_HOST}:${UPSTREAM_PORT};
|
||||
}
|
||||
|
||||
server {
|
||||
listen 80;
|
||||
server_name ${FQDN};
|
||||
return 301 https://$host$request_uri;
|
||||
}
|
||||
|
||||
server {
|
||||
listen 443 ssl;
|
||||
server_name ${FQDN};
|
||||
|
||||
ssl_certificate /etc/nginx/certs/live/${DOMAIN}/fullchain.pem;
|
||||
ssl_certificate_key /etc/nginx/certs/live/${DOMAIN}/privkey.pem;
|
||||
|
||||
# No `resolver`, and no `set $var` indirection — deliberately. Those exist so
|
||||
# nginx starts when a CONTAINER is absent; this upstream is a literal address
|
||||
# on the overlay, which needs no DNS at all. The three properties are one
|
||||
# decision, and placement is what decides them.
|
||||
location / {
|
||||
proxy_pass http://${NAME}_backend;
|
||||
proxy_set_header Host $host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $scheme;
|
||||
}
|
||||
}
|
||||
210
berth/ctrl/selftest.sh
Normal file
210
berth/ctrl/selftest.sh
Normal file
@@ -0,0 +1,210 @@
|
||||
#!/usr/bin/env bash
|
||||
# What berth has settled, and what it has withdrawn, written down as assertions.
|
||||
#
|
||||
# Two halves:
|
||||
# - decisions that hold. Failing one means "you are about to undo this".
|
||||
# - every entry in ../STALE.md. Failing one means a withdrawn assumption came
|
||||
# back. That is the half that makes STALE.md an audit surface and not an
|
||||
# archive — a retraction nobody re-reads is a retraction that decays.
|
||||
#
|
||||
# Scope: no cloud, no ssh, no sudo, no network. Cheap enough to actually run.
|
||||
# `make check` reports on the world and never fails; this exits 1, like rig's.
|
||||
#
|
||||
# Usage: make selftest (or: bash ctrl/selftest.sh)
|
||||
set -uo pipefail # NOT -e: one failing check must not abort the rest
|
||||
cd "$(dirname "$0")"
|
||||
|
||||
source ./lib/config.sh
|
||||
|
||||
rc=0
|
||||
passed=0
|
||||
check() { # name, expected, actual
|
||||
if [ "$2" = "$3" ]; then
|
||||
printf ' ok %s\n' "$1"
|
||||
passed=$((passed + 1))
|
||||
else
|
||||
printf ' FAIL %s\n expected: %s\n got: %s\n' "$1" "$2" "$3"
|
||||
rc=1
|
||||
fi
|
||||
}
|
||||
note() { printf '\n%s\n' "$1"; }
|
||||
skip() { printf ' skip %s (%s)\n' "$1" "$2"; }
|
||||
|
||||
# An absence check must not match the files that RECORD the absence. STALE.md
|
||||
# names every withdrawn thing by definition, and this file names them again to
|
||||
# assert them — so both are excluded, or every check fails on itself. rig hits
|
||||
# the same wall and assembles its pattern from fragments for the same reason.
|
||||
NOSELF="--exclude=selftest.sh --exclude=STALE.md"
|
||||
absent() { grep -rIl $NOSELF "$@" 2>/dev/null | wc -l; }
|
||||
|
||||
# A throwaway estate, for the checks that have to run berth rather than read it.
|
||||
TMP_ESTATE=_selftest
|
||||
cleanup() { rm -f "../estate/${TMP_ESTATE}.json"; }
|
||||
trap cleanup EXIT
|
||||
|
||||
note "the withdrawn assumptions — ../STALE.md, one check each"
|
||||
|
||||
# B1 — Pulumi. The two surviving mentions are historical fact about ppl/infra
|
||||
# and live in README.md and the estate, not in anything that runs.
|
||||
check "B1 no pulumi in the code" "0" "$(absent -i pulumi . ../Makefile)"
|
||||
|
||||
# B2 — the ctlptl precedent. Withdrawn; the argument stands on its own now.
|
||||
check "B2 the withdrawn precedent is cited nowhere" "0" "$(absent -i ctlptl ..)"
|
||||
|
||||
# B3 — `wg show <if> dump` leaks the private key in field 1. Stating only
|
||||
# show-vs-showconf makes the dump form read as safe.
|
||||
check "B3 all three wg forms are named" "yes" \
|
||||
"$(grep -q 'dump' vpn.sh && grep -q 'showconf' vpn.sh && echo yes || echo no)"
|
||||
check "B3 capture refuses showconf-shaped input" "1" \
|
||||
"$(printf '[Interface]\nPrivateKey = x\n' | bash vpn.sh capture >/dev/null 2>&1; echo $?)"
|
||||
check "B3 capture refuses dump-shaped input" "1" \
|
||||
"$(printf 'priv\tpub\t51820\toff\n' | bash vpn.sh capture >/dev/null 2>&1; echo $?)"
|
||||
|
||||
# B4 — keepalive belongs to the peer that DIALS, not the one that roams. The
|
||||
# first version warned on a correctly configured overlay, so the check is run
|
||||
# against one: hub carries the keepalive, nrft roams.
|
||||
python3 - <<'PY'
|
||||
import json, collections
|
||||
d = json.load(open("../estate/mcrn.json"), object_pairs_hook=collections.OrderedDict)
|
||||
d["vpn"]["overlays"]["estate"]["peers"]["box"]["keepalive"] = 25
|
||||
json.dump(d, open("../estate/_selftest.json", "w"), indent=2, ensure_ascii=False)
|
||||
PY
|
||||
check "B4 a correct overlay raises no keepalive warning" "0" \
|
||||
"$(ESTATE=$TMP_ESTATE bash vpn.sh check 2>/dev/null | grep -ci 'no peer entry carries')"
|
||||
|
||||
# B6 — public keys are 44-char base64 too, so shape alone would flag correct
|
||||
# data. Same fixture, with a real-shaped public key on a peer.
|
||||
python3 - <<'PY'
|
||||
import base64, collections, json, os
|
||||
d = json.load(open("../estate/_selftest.json"), object_pairs_hook=collections.OrderedDict)
|
||||
d["vpn"]["overlays"]["estate"]["peers"]["box"]["public_key"] = base64.b64encode(os.urandom(32)).decode()
|
||||
json.dump(d, open("../estate/_selftest.json", "w"), indent=2, ensure_ascii=False)
|
||||
PY
|
||||
check "B6 a public key does not trip the secret check" "0" \
|
||||
"$(ESTATE=$TMP_ESTATE bash vpn.sh check 2>/dev/null | grep -c 'FAIL.*key')"
|
||||
|
||||
cleanup # the fixture is done with; two estate files would make load_config
|
||||
# refuse to guess below, which is right but reads as a config failure
|
||||
|
||||
# B5 — the pass-through block must be LAST, or a subcommand that names a real
|
||||
# target runs that target too. Checked through make, not by reading the file.
|
||||
note "B5 a subcommand that names a target dispatches once"
|
||||
for combo in "host ports" "host services" "vpn check" "vpn show estate" "estate show"; do
|
||||
check " make $combo" "1" \
|
||||
"$(cd .. && make -n $combo 2>/dev/null | grep -c 'bash ctrl/')"
|
||||
done
|
||||
|
||||
# B7 — the overlay moved out of network.wireguard into a top-level vpn block.
|
||||
check "B7 nothing reads network.wireguard" "0" "$(absent 'network\.wireguard' .)"
|
||||
|
||||
# B8 — peers, not relatives. berth sources nothing from rig.
|
||||
check "B8 berth sources nothing from rig" "0" "$(absent -E 'rig/ctrl|\.\./rig' .)"
|
||||
|
||||
# B9 — langfuse was filed as an exception a template could not express. It was
|
||||
# the general case. The proof is a live route: render it and diff against the
|
||||
# hand-written file, normalised for comments and whitespace.
|
||||
LIVE=/home/mariano/wdir/semester/ppl/gateway/nginx/conf.d/langfuse.conf
|
||||
if [ -f "$LIVE" ]; then
|
||||
norm() { sed -e 's/#.*//' -e 's/[[:space:]]\+/ /g' -e 's/^ //' -e 's/ $//' -e '/^$/d' "$1"; }
|
||||
bash services.sh render aws >/dev/null 2>&1
|
||||
check "B9 the generated vhost reproduces the live one" "same" \
|
||||
"$(diff -q <(norm ./render/out/aws/langfuse.conf) <(norm "$LIVE") >/dev/null 2>&1 \
|
||||
&& echo same || echo different)"
|
||||
else
|
||||
skip "B9 generated vhost matches the live one" "ppl not on this machine"
|
||||
fi
|
||||
|
||||
note "the safety contract — berth's verbs are not all safe"
|
||||
|
||||
check "estate defaults to show" "show" "$(cd .. && make -n estate 2>/dev/null | grep -oE 'estate\.sh [a-z]+' | awk '{print $2}')"
|
||||
check "certs defaults to status" "status" "$(cd .. && make -n certs 2>/dev/null | grep -oE 'certs\.sh [a-z]+' | awk '{print $2}')"
|
||||
check "dns defaults to list" "list" "$(cd .. && make -n dns 2>/dev/null | grep -oE 'dns\.sh [a-z]+' | awk '{print $2}')"
|
||||
check "vpn defaults to list" "list" "$(cd .. && make -n vpn 2>/dev/null | grep -oE 'vpn\.sh [a-z]+' | awk '{print $2}')"
|
||||
|
||||
for verb in apply destroy; do
|
||||
check "estate $verb refuses without --yes" "1" \
|
||||
"$(bash estate.sh "$verb" >/dev/null 2>&1; echo $?)"
|
||||
done
|
||||
for verb in renew push; do
|
||||
check "certs $verb refuses" "1" \
|
||||
"$(bash certs.sh "$verb" >/dev/null 2>&1; echo $?)"
|
||||
done
|
||||
check "dns add refuses to change live DNS" "1" \
|
||||
"$(bash dns.sh add selftest >/dev/null 2>&1; echo $?)"
|
||||
check "vpn up refuses without --yes" "1" \
|
||||
"$(bash vpn.sh up >/dev/null 2>&1; echo $?)"
|
||||
|
||||
note "config — the caller's env beats the files"
|
||||
|
||||
# Generated from CONFIG_OVERRIDABLE, so a new key enrols itself.
|
||||
test_value() {
|
||||
case "$1" in
|
||||
TARGET) echo "gcp" ;;
|
||||
ESTATE) echo "mcrn" ;;
|
||||
*) echo "selftest-sentinel" ;;
|
||||
esac
|
||||
}
|
||||
for key in $CONFIG_OVERRIDABLE; do
|
||||
want="$(test_value "$key")"
|
||||
got="$(export "$key=$want"; load_config >/dev/null 2>&1; echo "${!key}")"
|
||||
check " caller's $key wins" "$want" "$got"
|
||||
done
|
||||
|
||||
note "rig agreement — recomputed, never imported"
|
||||
|
||||
# rig pins these same constants in its own selftest. Both arrive at them from
|
||||
# the same formula with no shared code, which is the coupling rule made testable.
|
||||
check "derive_port_base rig" "20310" "$(derive_port_base rig)"
|
||||
check "derive_port_base foo" "21690" "$(derive_port_base foo)"
|
||||
check "derive_port_base my-proj" "21030" "$(derive_port_base my-proj)"
|
||||
|
||||
note "containment — berth writes nothing outside berth/"
|
||||
|
||||
check "no tracked change outside berth/" "0" \
|
||||
"$(cd ../.. && git status --porcelain 2>/dev/null | grep -vc '^.. berth/')"
|
||||
check "generated output is ignored" "yes" \
|
||||
"$(cd .. && git check-ignore -q ctrl/render/out && echo yes || echo no)"
|
||||
# A trailing-slash pattern matches directories only, so ask about a path
|
||||
# inside it rather than the (not-yet-existing) directory itself.
|
||||
check "key material is ignored" "yes" \
|
||||
"$(cd .. && git check-ignore -q ctrl/.secrets/vpn/any.key && echo yes || echo no)"
|
||||
|
||||
note "capture and the checks that read it — three bugs found by running, 2026-09-14"
|
||||
|
||||
# 1. A placed service's upstream is DERIVED (✖ B9). Anything reading the raw
|
||||
# `upstream` field sees "" and skips it — which is how vpn.sh's bindings
|
||||
# invariant, the "my configurations broke" detector, went quiet the day
|
||||
# placement landed while still printing OK. Vacuous passes are the failure
|
||||
# mode this whole file exists to catch.
|
||||
check "a placed service resolves to a real upstream" "10.8.0.2:3000" \
|
||||
"$(bash -c 'source ./lib/config.sh; source ./lib/estate.sh; load_config >/dev/null;
|
||||
service_upstream "" local nrft 3000')"
|
||||
check "bindings actually inspects a service" "1" \
|
||||
"$(bash ./vpn.sh check 2>/dev/null | grep -c 'no service currently has an overlay address' \
|
||||
| awk '{print 1-$1}')"
|
||||
|
||||
# 2. A listen port belongs to a PEER. The roaming peer's is an ephemeral source
|
||||
# port; writing it to the overlay renames the port the firewall rule is
|
||||
# checked against — silently, since both are plausible integers.
|
||||
check "a roaming peer's port is not the overlay's port" "51820" \
|
||||
"$(python3 -c 'import json;print(json.load(open("../estate/mcrn.json"))["vpn"]["overlays"]["estate"]["listen_port"])')"
|
||||
|
||||
# 3. _status is always non-empty — capture rewrites it rather than clearing it —
|
||||
# so a warning gated on "is it set" can never turn off, including after the
|
||||
# capture it asks for. Gate on the structure instead.
|
||||
check "the capture warning clears once keys are in" "0" \
|
||||
"$(bash ./check.sh 2>/dev/null | grep -c 'public keys not captured')"
|
||||
|
||||
note "every STALE entry has a check here"
|
||||
|
||||
# Not "$0": line 15 cd's into this script's directory, so a relative $0 no
|
||||
# longer resolves. After the cd the file is simply selftest.sh.
|
||||
# Ids are counted wherever they appear — B5's sits in a note(), not a check name.
|
||||
entries="$(grep -c '^\*\*✖ B' ../STALE.md)"
|
||||
checked="$(grep -oE '\bB[1-9][0-9]?\b' selftest.sh | sort -u | wc -l)"
|
||||
check "STALE.md entries are all covered" "$entries" "$checked"
|
||||
|
||||
printf '\n%d passed' "$passed"
|
||||
[ "$rc" -ne 0 ] && printf ', SOME FAILED'
|
||||
printf '\n'
|
||||
exit "$rc"
|
||||
183
berth/ctrl/services.sh
Normal file
183
berth/ctrl/services.sh
Normal file
@@ -0,0 +1,183 @@
|
||||
#!/usr/bin/env bash
|
||||
# Gateway routes, projected from the estate onto one target.
|
||||
#
|
||||
# Usage:
|
||||
# ./services.sh # list
|
||||
# ./services.sh render aws # -> render/out/aws/*.conf (nginx vhosts)
|
||||
# ./services.sh render local # -> render/out/local/Caddyfile
|
||||
# ./services.sh deploy # refuses; ppl/ctrl/deploy.sh ships config
|
||||
#
|
||||
# Each target is a projection with its own rules, not a format conversion.
|
||||
# The nine axes they disagree on, and the install order: ../README.md
|
||||
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")"
|
||||
|
||||
source ./lib/config.sh
|
||||
source ./lib/estate.sh
|
||||
load_config
|
||||
|
||||
OUT_ROOT="./render/out"
|
||||
|
||||
list() {
|
||||
printf '%-12s %-16s %-22s %-9s %s\n' NAME FQDN UPSTREAM PLACEMENT SOURCE
|
||||
local name host up kind raw
|
||||
while IFS=$'\x1f' read -r name host up kind raw placement peer port lhost; do
|
||||
[ -z "$name" ] && continue
|
||||
# A placed service has no literal `upstream` — it is derived from the
|
||||
# peer's overlay address, so show what it actually resolves to.
|
||||
local shown; shown="$(service_upstream "$up" "$placement" "$peer" "$port")" || shown=""
|
||||
printf '%-12s %-16s %-22s %-9s %s\n' \
|
||||
"$name" "${host}.${DOMAIN}" "${shown:--}" "$placement" \
|
||||
"$([ -n "$raw" ] && echo 'hand-written' || echo 'generated')"
|
||||
done < <(estate_services "$TARGET")
|
||||
echo
|
||||
echo "hand-written entries are NOT generated and NOT overwritten."
|
||||
echo "run 'make estate show' to see why each one is an exception."
|
||||
}
|
||||
|
||||
render_cloud() {
|
||||
# Two statements: `local a="$1" b="$a"` expands all arguments before any
|
||||
# assignment, so $a would still be unset.
|
||||
local target="$1"
|
||||
local out="$OUT_ROOT/$target"
|
||||
rm -rf "$out"; mkdir -p "$out"
|
||||
local name host up kind raw uhost uport n=0 skipped=0
|
||||
|
||||
while IFS=$'\x1f' read -r name host up kind raw placement peer port lhost; do
|
||||
[ -z "$name" ] && continue
|
||||
if [ -n "$raw" ]; then
|
||||
skipped=$((skipped + 1))
|
||||
continue
|
||||
fi
|
||||
# Placement picks the rendering. A container on the estate's own network
|
||||
# is reached by NAME through docker's resolver; anything on the overlay
|
||||
# is reached by ADDRESS and needs no DNS. That is one decision, not the
|
||||
# three properties (upstream{}, no resolver, no set $var) it produces.
|
||||
local tmpl
|
||||
case "$placement" in
|
||||
local|instance)
|
||||
tmpl=./render/nginx-upstream.tmpl
|
||||
uhost="$(overlay_get estate "peers.${peer}.address")"
|
||||
uport="$port" # resolved via service_upstream's same rule
|
||||
if [ -z "$uhost" ]; then
|
||||
echo " ! $name: placement '$placement' names peer '$peer', which has no address" >&2
|
||||
continue
|
||||
fi
|
||||
;;
|
||||
hosted)
|
||||
echo " ! $name: placement 'hosted' is declared but not rendered yet" >&2
|
||||
continue
|
||||
;;
|
||||
*)
|
||||
if [ "$kind" = "static" ]; then
|
||||
tmpl=./render/nginx-static.tmpl
|
||||
uhost=""; uport=""
|
||||
else
|
||||
tmpl=./render/nginx-proxy.tmpl
|
||||
uhost="${up%%:*}"; uport="${up##*:}"
|
||||
fi
|
||||
;;
|
||||
esac
|
||||
sed -e "s|\${NAME}|${name}|g" \
|
||||
-e "s|\${ESTATE}|${ESTATE}|g" \
|
||||
-e "s|\${FQDN}|${host}.${DOMAIN}|g" \
|
||||
-e "s|\${DOMAIN}|${DOMAIN}|g" \
|
||||
-e "s|\${PLACEMENT}|${placement}|g" \
|
||||
-e "s|\${PEER}|${peer}|g" \
|
||||
-e "s|\${UPSTREAM_HOST}|${uhost}|g" \
|
||||
-e "s|\${UPSTREAM_PORT}|${uport}|g" \
|
||||
"$tmpl" > "$out/${name}.conf"
|
||||
n=$((n + 1))
|
||||
done < <(estate_services "$target")
|
||||
|
||||
echo "wrote $n vhost(s) to $out/ ($skipped hand-written, left alone)"
|
||||
cat <<EONOTE
|
||||
|
||||
TO INSTALL THESE, THREE THINGS MUST HAPPEN IN THIS ORDER — and the order is the
|
||||
whole reason this is not a one-liner:
|
||||
|
||||
1. These land in conf.d/generated/, NOT conf.d/. ppl/ctrl/deploy.sh rsyncs the
|
||||
gateway with --delete; generated and hand-written config sharing one
|
||||
directory means one of them gets erased.
|
||||
|
||||
2. nginx.conf needs a THIRD include line. Its conf.d/*.conf glob does not
|
||||
recurse — which is exactly why conf.d/soleprint/*.conf already needs its
|
||||
own line at nginx.conf:28-30.
|
||||
|
||||
3. That new include changes LOAD ORDER, and load order decides which :443
|
||||
block catches unmatched names. So default.conf's commented-out
|
||||
':443 default_server' must be restored FIRST. 'make check' fails on this
|
||||
today, deliberately — it is a gate, not a warning.
|
||||
EONOTE
|
||||
}
|
||||
|
||||
render_local() {
|
||||
local out="$OUT_ROOT/local"; mkdir -p "$out"
|
||||
local ld; ld=$(estate_get "local_domain")
|
||||
: "${ld:=local.ar}"
|
||||
local name host up kind raw port drift=0
|
||||
|
||||
{
|
||||
cat <<EOH
|
||||
# GENERATED by berth from estate/${ESTATE}.json. Do not edit.
|
||||
# Regenerate: make services render local
|
||||
#
|
||||
# Install: sudo ln -sf \$PWD/Caddyfile /etc/caddy/Caddyfile && sudo systemctl reload caddy
|
||||
# All *.${ld} resolve to 127.0.0.1 via dnsmasq.
|
||||
#
|
||||
# Every site address carries an explicit :80. Without it Caddy 2 defaults to
|
||||
# :443 with auto-HTTPS, which on *.${ld} means cert provisioning attempts that
|
||||
# fail and break the listener. Plain HTTP only on this host.
|
||||
#
|
||||
# Caddy matches the MOST SPECIFIC site address, not the first — the opposite of
|
||||
# nginx, which matches exactly and otherwise falls to default_server. A name set
|
||||
# that is unambiguous here can be ambiguous on the box.
|
||||
EOH
|
||||
while IFS=$'\x1f' read -r name host up kind raw placement peer port lhost; do
|
||||
[ -z "$name" ] && continue
|
||||
port=$(python3 -c '
|
||||
import json,sys
|
||||
d=json.load(open(sys.argv[1]))
|
||||
for s in d.get("services",[]):
|
||||
if s.get("name")==sys.argv[2]:
|
||||
print(s.get("local_port") or ""); break
|
||||
' "$ESTATE_FILE" "$name")
|
||||
[ -z "$port" ] && continue
|
||||
echo
|
||||
echo "${lhost}.${ld}:80, *.${lhost}.${ld}:80 {"
|
||||
echo " reverse_proxy localhost:${port}"
|
||||
echo "}"
|
||||
done < <(estate_services local)
|
||||
} > "$out/Caddyfile"
|
||||
|
||||
echo "wrote $out/Caddyfile"
|
||||
echo
|
||||
echo "rig is not consulted and does not know this exists — its handover"
|
||||
echo "scrub refuses the string '${ld}'. Where a port belongs to a rig,"
|
||||
echo "'make ports verify' RECOMPUTES rig's formula to check it rather than"
|
||||
echo "importing rig's code. Convention, verified; not a dependency."
|
||||
}
|
||||
|
||||
case "${1:-list}" in
|
||||
list) list ;;
|
||||
render)
|
||||
shift
|
||||
# `case "${1:-X}"` defaults the match but leaves $1 empty.
|
||||
t="${1:-$TARGET}"
|
||||
case "$t" in
|
||||
local) render_local ;;
|
||||
aws|gcp) render_cloud "$t" ;;
|
||||
*) echo "usage: $0 render [aws|gcp|local]" >&2; exit 1 ;;
|
||||
esac
|
||||
;;
|
||||
deploy)
|
||||
echo "berth does not ship config; ppl/ctrl/deploy.sh does." >&2
|
||||
echo " berth's half is the DESCRIPTION and the render. Shipping is" >&2
|
||||
echo " rsync + compose against a live box, and it belongs where the" >&2
|
||||
echo " credentials are: berth is the tool, ppl is the estate that" >&2
|
||||
echo " holds the secrets." >&2
|
||||
exit 1
|
||||
;;
|
||||
*) echo "usage: $0 [list|render [aws|gcp|local]|deploy]" >&2; exit 1 ;;
|
||||
esac
|
||||
11
berth/ctrl/versions.env
Normal file
11
berth/ctrl/versions.env
Normal file
@@ -0,0 +1,11 @@
|
||||
# Pinned toolchain. Committed. The weakest config layer.
|
||||
|
||||
# The infra executor. One, not a pair — see README.md.
|
||||
# This pin is a placeholder; set it from `tofu version` once installed.
|
||||
TOFU_VERSION=1.9.0
|
||||
TOFU_BIN=tofu
|
||||
|
||||
# certbot runs as a throwaway container so the DNS plugin's credentials never
|
||||
# have to be installed on this machine.
|
||||
CERTBOT_AWS_IMAGE=certbot/dns-route53:latest
|
||||
CERTBOT_GCP_IMAGE=certbot/dns-google:latest
|
||||
478
berth/ctrl/vpn.sh
Normal file
478
berth/ctrl/vpn.sh
Normal file
@@ -0,0 +1,478 @@
|
||||
#!/usr/bin/env bash
|
||||
# Overlays — WireGuard as berth's network layer.
|
||||
#
|
||||
# Usage:
|
||||
# ./vpn.sh # list
|
||||
# ./vpn.sh show <overlay> # topology
|
||||
# ./vpn.sh check # invariants
|
||||
# ./vpn.sh render <peer> # peer's wg0.conf -> render/out/vpn/
|
||||
# ./vpn.sh keygen <peer> # keypair -> .secrets/; prints only the public key
|
||||
# ./vpn.sh up|down --yes # refuses; the host operates its own tunnel
|
||||
#
|
||||
# sudo wg show | ./vpn.sh capture [--write]
|
||||
#
|
||||
# `wg show` is the only safe form: `wg show <if> dump` puts the private key in
|
||||
# field 1, and `wg showconf` prints it outright. capture refuses both.
|
||||
#
|
||||
# Rationale, topology and key handling: ../README.md
|
||||
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")"
|
||||
|
||||
source ./lib/config.sh
|
||||
source ./lib/estate.sh
|
||||
load_config
|
||||
|
||||
SECRETS_DIR="./.secrets/vpn"
|
||||
OUT_DIR="./render/out/vpn"
|
||||
|
||||
WORST=0
|
||||
note() { echo " $*"; }
|
||||
warn() { echo " WARN $*"; [ "$WORST" -lt 1 ] && WORST=1; return 0; }
|
||||
bad() { echo " FAIL $*"; WORST=2; return 0; }
|
||||
|
||||
# Which addresses belong to THIS machine, so checks can distinguish what they
|
||||
# can actually see from what needs capturing elsewhere.
|
||||
my_overlay_addrs() { ip -4 -o addr show 2>/dev/null | awk '{split($4,a,"/"); print a[1]}'; }
|
||||
is_me() { my_overlay_addrs | grep -qxF "$1"; }
|
||||
|
||||
list() {
|
||||
local n sub port peers
|
||||
for n in $(overlay_names); do
|
||||
sub=$(overlay_get "$n" subnet)
|
||||
port=$(overlay_get "$n" listen_port)
|
||||
peers=$(overlay_peers "$n" | grep -c . || true)
|
||||
printf '%-10s %-16s port %-7s %s peer(s)\n' "$n" "$sub" "$port" "$peers"
|
||||
note "$(overlay_get "$n" purpose)"
|
||||
done
|
||||
local st; st=$(estate_get "vpn._status")
|
||||
[ -n "$st" ] && { echo; echo " !! $st"; }
|
||||
}
|
||||
|
||||
show() {
|
||||
local ov="${1:-}"
|
||||
[ -z "$ov" ] && { echo "usage: $0 show <overlay>" >&2; exit 1; }
|
||||
overlay_names | grep -qxF "$ov" || {
|
||||
echo "no such overlay: $ov" >&2
|
||||
echo "available: $(overlay_names | tr '\n' ' ')" >&2; exit 1; }
|
||||
|
||||
echo "overlay: $ov subnet $(overlay_get "$ov" subnet) udp/$(overlay_get "$ov" listen_port)"
|
||||
echo
|
||||
printf '%-8s %-12s %-9s %-22s %s\n' PEER ADDRESS ROLE ENDPOINT PUBKEY
|
||||
local name addr role ep pk aips ka
|
||||
while IFS=$'\x1f' read -r name addr role ep pk aips ka; do
|
||||
[ -z "$name" ] && continue
|
||||
printf '%-8s %-12s %-9s %-22s %s%s\n' \
|
||||
"$name" "$addr" "$role" "${ep:-—}" "${pk:-—}" \
|
||||
"$(is_me "$addr" && echo ' <- this machine')"
|
||||
done < <(overlay_peers "$ov")
|
||||
}
|
||||
|
||||
check() {
|
||||
local ov name addr role ep pk aips ka
|
||||
for ov in $(overlay_names); do
|
||||
local sub port
|
||||
sub=$(overlay_get "$ov" subnet); port=$(overlay_get "$ov" listen_port)
|
||||
echo "overlay '$ov' — $sub udp/$port"
|
||||
|
||||
# 1. addresses: unique, and inside the subnet. Two peers sharing an
|
||||
# address is a silent misroute, never an error message.
|
||||
local addrs; addrs=$(overlay_peers "$ov" | cut -d$'\x1f' -f2 | grep -v '^$' || true)
|
||||
local dupes; dupes=$(echo "$addrs" | sort | uniq -d)
|
||||
[ -n "$dupes" ] && bad "duplicate peer addresses: $(echo "$dupes" | tr '\n' ' ')"
|
||||
while IFS= read -r a; do
|
||||
[ -z "$a" ] && continue
|
||||
addr_in_subnet "$a" "$sub" || bad "$a is outside $sub"
|
||||
done <<< "$addrs"
|
||||
|
||||
while IFS=$'\x1f' read -r name addr role ep pk aips ka; do
|
||||
[ -z "$name" ] && continue
|
||||
|
||||
# AllowedIPs is cryptokey routing — route table and ACL at once.
|
||||
case "$aips" in
|
||||
*0.0.0.0/0*) warn "$name: AllowedIPs includes 0.0.0.0/0 — full-tunnel. Deliberate?" ;;
|
||||
esac
|
||||
|
||||
# A peer with no endpoint cannot be dialed; it must initiate.
|
||||
if [ -z "$ep" ] && [ "$role" != "roaming" ]; then
|
||||
warn "$name: role '$role' but no endpoint — nothing can dial it."
|
||||
fi
|
||||
# Keepalive is NOT on the roaming peer's own entry — it is set on
|
||||
# the entry for the peer it dials. Checked per-overlay below.
|
||||
[ -z "$pk" ] && note "$name: public_key not captured yet"
|
||||
done < <(overlay_peers "$ov")
|
||||
|
||||
# If anything roams, some peer entry must carry a keepalive.
|
||||
if overlay_peers "$ov" | cut -d$'\x1f' -f3 | grep -qx roaming; then
|
||||
if ! overlay_peers "$ov" | cut -d$'\x1f' -f7 | grep -qE '^[0-9]+$'; then
|
||||
warn "a peer roams but no peer entry carries PersistentKeepalive"
|
||||
note " the roaming side sets it on the entry for the peer it dials;"
|
||||
note " without it the NAT mapping expires and the tunnel works only"
|
||||
note " while traffic flows outward — 'works sometimes'"
|
||||
else
|
||||
note "keepalive present on the dialed peer."
|
||||
fi
|
||||
fi
|
||||
|
||||
# 4. the listen port must be open wherever a peer is dialable.
|
||||
local fwports; fwports=$(estate_get "firewall" | python3 -c '
|
||||
import json,sys
|
||||
try: print(" ".join(str(r.get("port")) for r in json.load(sys.stdin)))
|
||||
except Exception: pass')
|
||||
case " $fwports " in
|
||||
*" $port "*) note "udp/$port present in the firewall description." ;;
|
||||
*) bad "udp/$port is in no firewall rule — no peer could be dialed." ;;
|
||||
esac
|
||||
echo
|
||||
done
|
||||
|
||||
# Public keys are also 44-char base64, so shape alone proves nothing. The
|
||||
# assertions are: no field named private, no key outside a public_key field.
|
||||
echo "secrets — the description must never carry a private key"
|
||||
local leaked
|
||||
leaked=$(python3 - estate/../../estate/*.json <<'PY' 2>/dev/null || true
|
||||
import json, re, sys, glob
|
||||
KEY = re.compile(r'^[A-Za-z0-9+/]{43}=$')
|
||||
bad = []
|
||||
for f in glob.glob("../estate/*.json"):
|
||||
def walk(node, path):
|
||||
if isinstance(node, dict):
|
||||
for k, v in node.items():
|
||||
if re.search(r'priv', k, re.I):
|
||||
bad.append(f"{f}: field '{'.'.join(path+[k])}' is named private")
|
||||
walk(v, path + [k])
|
||||
elif isinstance(node, list):
|
||||
for i, v in enumerate(node): walk(v, path + [str(i)])
|
||||
elif isinstance(node, str) and KEY.match(node):
|
||||
if not path or 'public' not in path[-1]:
|
||||
bad.append(f"{f}: base64 key at '{'.'.join(path)}' is not a public_key field")
|
||||
walk(json.load(open(f)), [])
|
||||
print("\n".join(bad))
|
||||
PY
|
||||
)
|
||||
if [ -n "$leaked" ]; then
|
||||
echo "$leaked" | while IFS= read -r l; do [ -n "$l" ] && bad "$l"; done
|
||||
else
|
||||
note "clean — no private-named field, no stray key material."
|
||||
fi
|
||||
echo
|
||||
|
||||
# A service reached over the overlay must bind an address the tunnel can
|
||||
# reach. Loopback cannot be reached through a tunnel.
|
||||
echo "bindings — services reached over the overlay must bind a reachable address"
|
||||
local checked=0
|
||||
while IFS=$'\x1f' read -r name host up kind raw placement peer port lhost; do
|
||||
# Resolve placement first: a placed service's upstream is derived, not
|
||||
# literal, so reading `up` alone skips it and this invariant goes quiet.
|
||||
up="$(service_upstream "$up" "$placement" "$peer" "$port")" || true
|
||||
[ -z "$up" ] && continue
|
||||
local uhost="${up%%:*}" uport="${up##*:}"
|
||||
addr_in_subnet "$uhost" "$(overlay_get estate subnet)" 2>/dev/null || continue
|
||||
checked=$((checked + 1))
|
||||
if is_me "$uhost"; then
|
||||
local binds; binds=$(ss -ltn 2>/dev/null | awk -v p=":$uport\$" '$4 ~ p {print $4}')
|
||||
if [ -z "$binds" ]; then
|
||||
bad "$name: nothing listens on :$uport here, but $uhost:$uport is its upstream"
|
||||
elif echo "$binds" | grep -q '^127\.0\.0\.1:'; then
|
||||
bad "$name: :$uport binds 127.0.0.1 — unreachable over the overlay"
|
||||
note " the tunnel cannot reach loopback; bind 0.0.0.0 or $uhost"
|
||||
else
|
||||
note "$name: :$uport binds $(echo "$binds" | tr '\n' ' ')— reachable"
|
||||
echo "$binds" | grep -q '^0\.0\.0\.0:' && \
|
||||
note " (0.0.0.0 also exposes it to the LAN; $uhost alone would be tighter)"
|
||||
fi
|
||||
else
|
||||
note "$name: upstream $uhost is another peer — needs capture there"
|
||||
fi
|
||||
done < <(estate_services "$TARGET")
|
||||
[ "$checked" = 0 ] && note "no service currently has an overlay address as its upstream."
|
||||
echo
|
||||
|
||||
case "$WORST" in
|
||||
0) echo "OK" ;;
|
||||
1) echo "OK, with warnings" ;;
|
||||
2) echo "PROBLEMS FOUND — see FAIL lines above" ;;
|
||||
esac
|
||||
return 0
|
||||
}
|
||||
|
||||
# A .gitignore pattern containing a slash anchors to its own directory, so the
|
||||
# only way to know a path is ignored is to ask git.
|
||||
assert_ignored() {
|
||||
local path="$1"
|
||||
if ! git check-ignore -q "$path" 2>/dev/null; then
|
||||
echo "REFUSING: '$path' is not gitignored." >&2
|
||||
echo " Writing key material there would stage it on the next 'git add'." >&2
|
||||
echo " Verify with: git check-ignore -v $path" >&2
|
||||
exit 1
|
||||
fi
|
||||
}
|
||||
|
||||
keygen() {
|
||||
local peer="${1:-}"
|
||||
[ -z "$peer" ] && { echo "usage: $0 keygen <peer>" >&2; exit 1; }
|
||||
command -v wg >/dev/null || { echo "wg not installed." >&2; exit 1; }
|
||||
|
||||
mkdir -p "$SECRETS_DIR"
|
||||
assert_ignored "$SECRETS_DIR"
|
||||
local kf="$SECRETS_DIR/${peer}.key"
|
||||
[ -e "$kf" ] && { echo "REFUSING: $kf exists. Delete it deliberately to rotate." >&2; exit 1; }
|
||||
|
||||
( umask 077; wg genkey > "$kf" )
|
||||
echo "private key -> $kf (0600, gitignored, never leaves this machine)"
|
||||
echo
|
||||
echo "public key for the estate description:"
|
||||
echo " $(wg pubkey < "$kf")"
|
||||
echo
|
||||
echo "Paste that into estate/*.json under vpn.overlays.<ov>.peers.${peer}.public_key."
|
||||
echo "The private key stays here and is injected only at render time."
|
||||
}
|
||||
|
||||
render() {
|
||||
local peer="${1:-}"
|
||||
[ -z "$peer" ] && { echo "usage: $0 render <peer>" >&2; exit 1; }
|
||||
mkdir -p "$OUT_DIR"
|
||||
assert_ignored "$OUT_DIR"
|
||||
|
||||
local ov=estate
|
||||
local found=""
|
||||
local name addr role ep pk aips ka
|
||||
while IFS=$'\x1f' read -r name addr role ep pk aips ka; do
|
||||
[ "$name" = "$peer" ] && found=1 && break
|
||||
done < <(overlay_peers "$ov")
|
||||
[ -z "$found" ] && { echo "no such peer '$peer' in overlay '$ov'" >&2; exit 1; }
|
||||
|
||||
local missing=""
|
||||
while IFS=$'\x1f' read -r name addr role ep pk aips ka; do
|
||||
[ -z "$pk" ] && missing="$missing $name"
|
||||
done < <(overlay_peers "$ov")
|
||||
if [ -n "$missing" ]; then
|
||||
echo "REFUSING to render: public keys not captured for:$missing" >&2
|
||||
echo " A config without every peer's public key is a config that silently" >&2
|
||||
echo " drops those peers. Capture first: sudo wg show | $0 capture --write" >&2
|
||||
exit 1
|
||||
fi
|
||||
echo "would write $OUT_DIR/${peer}.conf (all keys present)"
|
||||
}
|
||||
|
||||
# Reads `wg show` on stdin. Peers match by allowed-ips address, not public key,
|
||||
# because the keys are what is missing. A roaming peer's endpoint is a home
|
||||
# address and has no stable value — dropped in the parser, not just unused.
|
||||
capture() {
|
||||
local write="" as_peer=""
|
||||
while [ $# -gt 0 ]; do
|
||||
case "$1" in
|
||||
--write) write=1 ;;
|
||||
--as) shift; as_peer="${1:-}"
|
||||
[ -z "$as_peer" ] && { echo "--as needs a peer name" >&2; exit 1; } ;;
|
||||
*) echo "capture: unknown argument '$1'" >&2; exit 1 ;;
|
||||
esac
|
||||
shift
|
||||
done
|
||||
|
||||
local input; input=$(cat)
|
||||
if [ -z "$input" ]; then
|
||||
echo "nothing on stdin." >&2
|
||||
echo " run: sudo wg show | $0 capture" >&2
|
||||
exit 1
|
||||
fi
|
||||
# Refuse the unsafe forms outright rather than parsing around them.
|
||||
if printf '%s' "$input" | grep -qiE '^\s*PrivateKey\s*=|^\[Interface\]'; then
|
||||
echo "REFUSING: this looks like 'wg showconf' output — it contains a PRIVATE KEY." >&2
|
||||
echo " Use 'sudo wg show' (plain). It prints 'private key: (hidden)'." >&2
|
||||
exit 1
|
||||
fi
|
||||
if ! printf '%s' "$input" | grep -q 'interface:'; then
|
||||
echo "REFUSING: this does not look like 'wg show' output." >&2
|
||||
echo " If it was 'wg show <if> dump': that form's first field IS the" >&2
|
||||
echo " private key. Use 'sudo wg show' with no subcommand." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
WRITE="$write" AS_PEER="$as_peer" INPUT="$input" python3 - "$ESTATE_FILE" <<'PYCAP'
|
||||
import collections, ipaddress, json, os, re, sys
|
||||
|
||||
text = os.environ["INPUT"]
|
||||
write = os.environ.get("WRITE") == "1"
|
||||
path = sys.argv[1]
|
||||
|
||||
iface, peers, cur = {}, [], None
|
||||
for line in text.splitlines():
|
||||
st = line.strip()
|
||||
if st.startswith("interface:"):
|
||||
cur = iface; cur["name"] = st.split(":", 1)[1].strip(); continue
|
||||
if st.startswith("peer:"):
|
||||
cur = {"public_key": st.split(":", 1)[1].strip()}; peers.append(cur); continue
|
||||
if cur is None or ":" not in st:
|
||||
continue
|
||||
k, v = st.split(":", 1)
|
||||
k, v = k.strip().lower(), v.strip()
|
||||
if k == "private key":
|
||||
continue # never recorded, whatever it says
|
||||
if k == "public key": cur["public_key"] = v
|
||||
elif k == "listening port": cur["listen_port"] = v
|
||||
elif k == "allowed ips": cur["allowed_ips"] = v
|
||||
elif k == "endpoint": cur["endpoint"] = v
|
||||
elif k == "persistent keepalive":
|
||||
m = re.search(r"(\d+)", v)
|
||||
if m: cur["keepalive"] = int(m.group(1))
|
||||
|
||||
d = json.load(open(path), object_pairs_hook=collections.OrderedDict)
|
||||
ov = d["vpn"]["overlays"]["estate"]
|
||||
|
||||
# address -> peer name, from what the estate already declares
|
||||
by_addr = {p["address"]: n for n, p in ov["peers"].items() if p.get("address")}
|
||||
by_key = {p["public_key"]: n for n, p in ov["peers"].items() if p.get("public_key")}
|
||||
subnet = ipaddress.ip_network(ov["subnet"]) if ov.get("subnet") else None
|
||||
hubs = [n for n, p in ov["peers"].items() if p.get("role") == "hub"]
|
||||
|
||||
# Whose interface block is this? `--as` names it explicitly, and that is the only
|
||||
# thing that works for output captured over ssh: the addresses on THIS machine
|
||||
# say nothing about the machine the output came from.
|
||||
as_peer = os.environ.get("AS_PEER") or ""
|
||||
if as_peer:
|
||||
if as_peer not in ov["peers"]:
|
||||
print("no peer named %r in this overlay. known: %s"
|
||||
% (as_peer, ", ".join(ov["peers"])))
|
||||
raise SystemExit(1)
|
||||
me = as_peer
|
||||
else:
|
||||
me = None
|
||||
local = os.popen(
|
||||
"ip -4 -o addr show 2>/dev/null | awk '{split($4,a,\"/\"); print a[1]}'"
|
||||
).read().split()
|
||||
for n, p in ov["peers"].items():
|
||||
if p.get("address") and p["address"] in local:
|
||||
me = n
|
||||
|
||||
changes = []
|
||||
conflicts = []
|
||||
staged = {}
|
||||
def setf(peer, field, val, why=""):
|
||||
p = ov["peers"][peer]
|
||||
if val is None or p.get(field) == val:
|
||||
return
|
||||
# Two values for one field means the input is from another machine.
|
||||
prev = staged.get((peer, field))
|
||||
if prev is not None and prev != val:
|
||||
conflicts.append((peer, field, prev, val))
|
||||
return
|
||||
staged[(peer, field)] = val
|
||||
changes.append((peer, field, p.get(field), val, why))
|
||||
if write:
|
||||
p[field] = val
|
||||
|
||||
if me and iface.get("public_key"):
|
||||
setf(me, "public_key", iface["public_key"], "(this machine's interface)")
|
||||
|
||||
def match(pr):
|
||||
# 1. The public key IS the identity. Use it whenever the estate knows it.
|
||||
n = by_key.get(pr["public_key"])
|
||||
if n:
|
||||
return n
|
||||
nets = [a.strip() for a in pr.get("allowed_ips", "").split(",") if a.strip()]
|
||||
# 2. An allowed-ip that is a declared peer address — the ordinary spoke case.
|
||||
for a in nets:
|
||||
if a.split("/")[0] in by_addr:
|
||||
return by_addr[a.split("/")[0]]
|
||||
# 3. A peer routing the WHOLE overlay is the hub seen from a spoke. Its
|
||||
# allowed_ips is the subnet itself, so no single address ever matches it.
|
||||
if subnet and len(hubs) == 1:
|
||||
for a in nets:
|
||||
try:
|
||||
if ipaddress.ip_network(a, strict=False).supernet_of(subnet):
|
||||
return hubs[0]
|
||||
except ValueError:
|
||||
continue
|
||||
return None
|
||||
|
||||
for pr in peers:
|
||||
name = match(pr)
|
||||
if not name:
|
||||
changes.append(("?", "UNMATCHED", None,
|
||||
"allowed_ips=%s key=%s" % (pr.get("allowed_ips"), pr["public_key"][:12] + "..."),
|
||||
"no estate peer has this key, this address, or this route"))
|
||||
continue
|
||||
setf(name, "public_key", pr.get("public_key"))
|
||||
setf(name, "allowed_ips", pr.get("allowed_ips"))
|
||||
setf(name, "keepalive", pr.get("keepalive"))
|
||||
# endpoint: recorded ONLY for a non-roaming peer. For a roaming one the
|
||||
# value is a home ISP address and is deliberately dropped here.
|
||||
if pr.get("endpoint"):
|
||||
if ov["peers"][name].get("role") == "roaming":
|
||||
changes.append((name, "endpoint", None, "(dropped: roaming peer)",
|
||||
"a home address is the one sensitive field; roaming peers have no stable endpoint"))
|
||||
else:
|
||||
setf(name, "endpoint", pr["endpoint"])
|
||||
|
||||
if me and iface.get("listen_port"):
|
||||
try:
|
||||
lp = int(iface["listen_port"])
|
||||
except ValueError:
|
||||
lp = None
|
||||
if lp is not None:
|
||||
# A listen port belongs to the PEER, not to the overlay. A roaming peer's
|
||||
# is an ephemeral source port chosen by the kernel; writing it to the
|
||||
# overlay would rename the port the firewall rule is checked against.
|
||||
setf(me, "listen_port", lp)
|
||||
if ov["peers"][me].get("role") == "hub" and ov.get("listen_port") != lp:
|
||||
changes.append(("(overlay)", "listen_port", ov.get("listen_port"), lp,
|
||||
"the hub's port is the overlay's port"))
|
||||
if write:
|
||||
ov["listen_port"] = lp
|
||||
|
||||
if conflicts:
|
||||
print("REFUSING: the same field was reported twice with different values.\n")
|
||||
for peer, field, a, b in conflicts:
|
||||
print(" %s.%s: %s vs %s" % (peer, field, a, b))
|
||||
print("\nThis usually means `wg show` output from one machine was piped into")
|
||||
print("capture on another. Run capture on the machine the output came from.")
|
||||
raise SystemExit(1)
|
||||
|
||||
if not changes:
|
||||
print("nothing to record — the estate already matches what wg reports.")
|
||||
else:
|
||||
print("%-9s %-12s %-22s %s" % ("PEER", "FIELD", "WAS", "WOULD BE"))
|
||||
for peer, field, was, val, why in changes:
|
||||
print("%-9s %-12s %-22s %s" % (peer, field, was if was is not None else "—", val))
|
||||
if why: print(" %s" % why)
|
||||
|
||||
if write:
|
||||
still = [n for n, p in ov["peers"].items() if not p.get("public_key")]
|
||||
if not still:
|
||||
d["vpn"]["_status"] = ("CAPTURED %s — public keys, allowed-ips and keepalive read from "
|
||||
"`wg show`. Roaming endpoints deliberately not recorded."
|
||||
% __import__("datetime").date.today())
|
||||
json.dump(d, open(path, "w"), indent=2, ensure_ascii=False)
|
||||
open(path, "a").write("\n")
|
||||
print("\nwritten to %s" % path)
|
||||
else:
|
||||
print("\nnothing written. Add --write to record it.")
|
||||
PYCAP
|
||||
}
|
||||
|
||||
refuse() {
|
||||
local verb="$1"; shift
|
||||
local yes=""
|
||||
for a in "$@"; do [ "$a" = "--yes" ] && yes=1; done
|
||||
[ -z "$yes" ] && {
|
||||
echo "refusing to $verb without --yes." >&2
|
||||
echo " $verb changes live networking — it can cut the path this session" >&2
|
||||
echo " is reaching the estate through. Read 'make vpn check' first." >&2
|
||||
exit 1; }
|
||||
echo "refusing to $verb: not implemented. Bringing a tunnel up or down is" >&2
|
||||
echo " the host's business, and the live one is systemd-managed" >&2
|
||||
echo " (wg-quick@wg0). berth describes and renders; it does not operate." >&2
|
||||
exit 1
|
||||
}
|
||||
|
||||
case "${1:-list}" in
|
||||
list) list ;;
|
||||
show) shift; show "${1:-}" ;;
|
||||
check) check ;;
|
||||
render) shift; render "${1:-}" ;;
|
||||
keygen) shift; keygen "${1:-}" ;;
|
||||
capture) shift; capture "$@" ;;
|
||||
up|down) v="$1"; shift; refuse "$v" "$@" ;;
|
||||
*) echo "usage: $0 [list|show <ov>|check|render <peer>|keygen <peer>|capture [--as <peer>] [--write]|up --yes|down --yes]" >&2; exit 1 ;;
|
||||
esac
|
||||
326
berth/estate/mcrn.json
Normal file
326
berth/estate/mcrn.json
Normal file
@@ -0,0 +1,326 @@
|
||||
{
|
||||
"_meta": {
|
||||
"status": "UNVERIFIED — derived from the repos, not from the estate",
|
||||
"why": "ppl/infra/ was written and never applied: no ~/.pulumi, no infra/venv, no stack state, files dated 'mar 6'. The estate was built in the console and the IaC is aspirational. B1's inventory is what replaces these values with observed ones; until it runs, every field here is a CLAIM.",
|
||||
"never_record": "credential values. Resource ids and settings only. nova's gateway secret is deliberately absent from this file even though it is committed in plaintext in ppl/gateway/nginx/conf.d/nova.conf — see services[].raw.",
|
||||
"sources": [
|
||||
"ppl/infra/__main__.py",
|
||||
"ppl/ctrl/dns.sh",
|
||||
"ppl/ctrl/certs.sh",
|
||||
"ppl/gateway/docker-compose.yml",
|
||||
"ppl/gateway/nginx/conf.d/",
|
||||
"ppl/local/Caddyfile"
|
||||
],
|
||||
"placement": {
|
||||
"box": "a container on the estate's own docker network — upstream is the container name",
|
||||
"local": "a rig cluster on a peer, reached over the overlay — upstream is that peer's address",
|
||||
"instance": "a dedicated cloud instance on the overlay — same rendering as `local`",
|
||||
"hosted": "a managed endpoint. Declared so moving to one is a one-line change; unused.",
|
||||
"_why": "A service says WHERE it runs. How it is reached follows from that, and the three properties of a static-upstream vhost — upstream{}, no resolver, no set $var — are one decision rather than three."
|
||||
}
|
||||
},
|
||||
"domain": "mcrn.ar",
|
||||
"local_domain": "local.ar",
|
||||
"host": "mcrn",
|
||||
"host_admin": "mcrn-admin",
|
||||
"instance": {
|
||||
"type": "t3.small",
|
||||
"disk_gb": 30,
|
||||
"disk_type": "gp3",
|
||||
"image": "debian-12",
|
||||
"user": "mariano"
|
||||
},
|
||||
"firewall": [
|
||||
{
|
||||
"port": 22,
|
||||
"proto": "tcp",
|
||||
"desc": "SSH"
|
||||
},
|
||||
{
|
||||
"port": 80,
|
||||
"proto": "tcp",
|
||||
"desc": "HTTP"
|
||||
},
|
||||
{
|
||||
"port": 443,
|
||||
"proto": "tcp",
|
||||
"desc": "HTTPS"
|
||||
},
|
||||
{
|
||||
"port": 3022,
|
||||
"proto": "tcp",
|
||||
"desc": "Gitea SSH",
|
||||
"note": "compose maps 3022:22 but GITEA__server__SSH_PORT=22, so gitea advertises :22 in clone URLs while listening on :3022. B1 confirms which is real."
|
||||
},
|
||||
{
|
||||
"port": 51820,
|
||||
"proto": "udp",
|
||||
"desc": "WireGuard",
|
||||
"note": "ABSENT from ppl/infra/__main__.py's four rules — but the tunnel is live (ping 10.8.0.1 succeeds), so the real security group must already allow it. The code therefore does not describe the estate. Confirm in V1."
|
||||
}
|
||||
],
|
||||
"network": {
|
||||
"docker_network": "gateway",
|
||||
"docker_network_note": "A fixed, externally-joinable bridge name. Every unrelated app stack on the box joins it so nginx can resolve them by container name. This is why the gateway compose declares 8 services while nginx routes 20+ hostnames.",
|
||||
"wireguard_moved": "superseded by the top-level `vpn` block"
|
||||
},
|
||||
"vpn": {
|
||||
"_status": "CAPTURED 2026-09-14 — public keys, allowed-ips and keepalive read from `wg show`. Roaming endpoints deliberately not recorded.",
|
||||
"_never_record": "private keys. `wg show` prints 'private key: (hidden)' and is the safe capture command. `wg showconf` dumps PrivateKey= in clear — never use it.",
|
||||
"overlays": {
|
||||
"estate": {
|
||||
"purpose": "Connects the estate's machines across clouds without a shared VPC, and carries everything that does not need to be publicly reachable.",
|
||||
"subnet": "10.8.0.0/24",
|
||||
"listen_port": 51820,
|
||||
"peers": {
|
||||
"box": {
|
||||
"address": "10.8.0.1",
|
||||
"role": "hub",
|
||||
"note": "mcrn.ar. Has a public IP, so it is the peer others dial. Carries the registry (:5000) and woodpecker's gRPC (:9000), both bound to this address and therefore overlay-only.",
|
||||
"endpoint": "3.23.204.197:51820",
|
||||
"public_key": "zVYCmi3xucuX7k/aDhrOUPyN4GRk96ffSDD6dUFQjh4=",
|
||||
"allowed_ips": "10.8.0.0/24",
|
||||
"keepalive": 25,
|
||||
"listen_port": 51820
|
||||
},
|
||||
"nrft": {
|
||||
"address": "10.8.0.2",
|
||||
"role": "roaming",
|
||||
"note": "The dev box. Behind NAT, so it must initiate and needs PersistentKeepalive. Verified: wg0 UP at 10.8.0.2/24, ping 10.8.0.1 0% loss at 153ms.",
|
||||
"endpoint": null,
|
||||
"public_key": "zlIBGs4y5rt6uVdmFBasHpafht6ErxG+R3ySCg5rh3s=",
|
||||
"allowed_ips": "10.8.0.2/32, 192.168.1.0/24",
|
||||
"keepalive": null,
|
||||
"listen_port": 36145
|
||||
},
|
||||
"work": {
|
||||
"address": "10.8.0.3",
|
||||
"role": "roaming",
|
||||
"note": "A work computer, granted access when it was needed. Identified by the user at capture time, 2026-09-14 — it was NOT in the description before, and the wire is where it was found. No handshake and no transfer have ever been recorded for it, so it is a standing grant rather than a live peer: it can connect, and never has. Whether to keep or revoke it is the host's call.",
|
||||
"endpoint": null,
|
||||
"public_key": "ruSZwKt/p60GVsTLSAhcKBIXKkSZsf0gWSmSH1+UgE0=",
|
||||
"allowed_ips": "10.8.0.3/32",
|
||||
"keepalive": null
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"databases": [
|
||||
"gitea",
|
||||
"woodpecker",
|
||||
"umami"
|
||||
],
|
||||
"certs": {
|
||||
"issued": [
|
||||
"mcrn.ar",
|
||||
"*.mcrn.ar",
|
||||
"*.spr.mcrn.ar"
|
||||
],
|
||||
"issued_source": "ppl/ctrl/certs.sh:92 — the -d flags passed to certbot",
|
||||
"note": "What the cert ACTUALLY covers. estate_sans() derives what the services NEED. check.sh compares the two; the difference is the finding, not a restatement."
|
||||
},
|
||||
"services": [
|
||||
{
|
||||
"name": "gitea",
|
||||
"host": "git",
|
||||
"upstream": "gitea:3000",
|
||||
"targets": [
|
||||
"aws"
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "woodpecker",
|
||||
"host": "ci",
|
||||
"upstream": "woodpecker-server:8000",
|
||||
"targets": [
|
||||
"aws"
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "registry",
|
||||
"host": "registry",
|
||||
"upstream": "registry:5000",
|
||||
"targets": [
|
||||
"aws"
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "umami",
|
||||
"host": "analytics",
|
||||
"upstream": "umami:3000",
|
||||
"targets": [
|
||||
"aws"
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "docserve",
|
||||
"host": "docs",
|
||||
"upstream": "docserve:8020",
|
||||
"targets": [
|
||||
"aws"
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "ghost",
|
||||
"host": "notes",
|
||||
"upstream": "ghost:2368",
|
||||
"targets": [
|
||||
"aws"
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "deskmeter",
|
||||
"host": "deskmeter",
|
||||
"upstream": "dmweb:10000",
|
||||
"targets": [
|
||||
"aws"
|
||||
],
|
||||
"local_port": 10000
|
||||
},
|
||||
{
|
||||
"name": "sysmonstm",
|
||||
"host": "sysmonstm",
|
||||
"upstream": "sysmonstm-edge:8080",
|
||||
"targets": [
|
||||
"aws"
|
||||
],
|
||||
"local_port": 8020
|
||||
},
|
||||
{
|
||||
"name": "malvalava",
|
||||
"host": "malvalava",
|
||||
"upstream": "mlvclean-frontend:80",
|
||||
"targets": [
|
||||
"aws"
|
||||
],
|
||||
"local_port": 30090
|
||||
},
|
||||
{
|
||||
"name": "soleprint",
|
||||
"host": "soleprint",
|
||||
"upstream": "soleprint:8000",
|
||||
"targets": [
|
||||
"aws"
|
||||
],
|
||||
"local_port": 12000
|
||||
},
|
||||
{
|
||||
"name": "dlt",
|
||||
"host": "dlt.spr",
|
||||
"upstream": "dlt_spr:8000",
|
||||
"targets": [
|
||||
"aws"
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "sample",
|
||||
"host": "sample.spr",
|
||||
"upstream": "sample_spr:8000",
|
||||
"targets": [
|
||||
"aws"
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "mariano",
|
||||
"host": "mariano",
|
||||
"kind": "static",
|
||||
"targets": [
|
||||
"aws"
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "rigui",
|
||||
"host": "rig",
|
||||
"kind": "static",
|
||||
"targets": [
|
||||
"aws"
|
||||
],
|
||||
"local_port": 20310
|
||||
},
|
||||
{
|
||||
"name": "unt",
|
||||
"host": "unt",
|
||||
"targets": [
|
||||
"local"
|
||||
],
|
||||
"local_port": 8040
|
||||
},
|
||||
{
|
||||
"name": "mpr",
|
||||
"host": "mpr",
|
||||
"targets": [
|
||||
"local"
|
||||
],
|
||||
"local_port": 30080
|
||||
},
|
||||
{
|
||||
"name": "nvi",
|
||||
"host": "nvi",
|
||||
"targets": [
|
||||
"local"
|
||||
],
|
||||
"local_port": 8060
|
||||
},
|
||||
{
|
||||
"name": "eth",
|
||||
"host": "eth",
|
||||
"targets": [
|
||||
"local"
|
||||
],
|
||||
"local_port": 8050
|
||||
},
|
||||
{
|
||||
"name": "amar",
|
||||
"host": "amar",
|
||||
"targets": [
|
||||
"local"
|
||||
],
|
||||
"local_port": 8030
|
||||
},
|
||||
{
|
||||
"name": "nova",
|
||||
"host": "nova",
|
||||
"upstream": "nova-ui:80",
|
||||
"targets": [
|
||||
"aws"
|
||||
],
|
||||
"raw": true,
|
||||
"raw_why": "Gated on an X-Gateway-Secret header whose value is committed in plaintext. The value is NOT recorded here. Worse: stellarair.conf proxies to the SAME nova-ui:80 upstream WITHOUT the check, so the gate is bypassable by hostname. Stays hand-written until that is decided."
|
||||
},
|
||||
{
|
||||
"name": "stellarair",
|
||||
"host": "stellarair",
|
||||
"upstream": "nova-ui:80",
|
||||
"targets": [
|
||||
"aws"
|
||||
],
|
||||
"raw": true,
|
||||
"raw_why": "See nova. Same upstream, no header gate."
|
||||
},
|
||||
{
|
||||
"name": "langfuse",
|
||||
"host": "langfuse",
|
||||
"local_host": "lng",
|
||||
"placement": "local",
|
||||
"peer": "nrft",
|
||||
"port": 3000,
|
||||
"targets": [
|
||||
"aws",
|
||||
"local"
|
||||
],
|
||||
"local_port": 3000,
|
||||
"note": "One service, one socket, two names. It was two entries with one flagged `raw`; placement is what made the exception expressible, so it is generated now."
|
||||
},
|
||||
{
|
||||
"name": "legacy",
|
||||
"host": "*.soleprint",
|
||||
"upstream": "soleprint:8000",
|
||||
"targets": [
|
||||
"aws"
|
||||
],
|
||||
"raw": true,
|
||||
"raw_why": "A regex server_name with a named capture plus sub_filter injection — not expressible as a template. ALSO BROKEN: its /api/, /admin/, /static/ and / blocks proxy to 127.0.0.1, i.e. inside the nginx container where nothing listens, so every legacy room 502s. Only /wrapper/ uses the correct container-name form."
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -8,13 +8,13 @@
|
||||
#
|
||||
# spr depends on rig, never the other way round. Building and deleting a cluster
|
||||
# is rig's job, so up and down hand straight to rig/ctrl/cluster.sh, carrying the
|
||||
# four things that make this cluster spr's rather than rig's defaults:
|
||||
# the things that make this cluster spr's rather than rig's defaults:
|
||||
#
|
||||
# CLUSTER=spr rooms deploy into the kind-spr context
|
||||
# KIND_CONFIG spr's own shape, which maps the rooms' gateway NodePorts
|
||||
# KIND_CONFIG spr's own kind config, which maps the rooms' gateway NodePorts
|
||||
# REGISTRY_MODE=none rooms load images straight into the node
|
||||
# PROFILE=minimal pinned here, so a change to rig's own ctrl/.env can never
|
||||
# quietly add addons to spr's cluster
|
||||
# PROFILE= ADDONS= set empty here, so rig's own ctrl/.env can never quietly
|
||||
# OVERLAY= pick a profile, an overlay or addons for spr's cluster
|
||||
#
|
||||
# status stays here: it answers a question about rooms, not about the cluster.
|
||||
set -e
|
||||
@@ -26,7 +26,9 @@ rig() {
|
||||
CLUSTER=spr \
|
||||
KIND_CONFIG="$SCRIPT_DIR/k8s/kind-config.yaml" \
|
||||
REGISTRY_MODE=none \
|
||||
PROFILE=minimal \
|
||||
PROFILE= \
|
||||
OVERLAY= \
|
||||
ADDONS= \
|
||||
bash "$RIG_CTRL/cluster.sh" "$@"
|
||||
}
|
||||
|
||||
|
||||
80
ctrl/theme.sh
Executable file
80
ctrl/theme.sh
Executable file
@@ -0,0 +1,80 @@
|
||||
#!/usr/bin/env bash
|
||||
# Bake the theme and its parts into the pages that use them.
|
||||
#
|
||||
# Usage:
|
||||
# ./ctrl/theme.sh # bake — rewrite every generated block
|
||||
# ./ctrl/theme.sh check # fail if any page is stale; changes nothing
|
||||
# ./ctrl/theme.sh new [title] # a scaffold page to start from
|
||||
# ./ctrl/theme.sh run FILE [--list|--check] [--only NAME]
|
||||
# # every page a run file lists, each with a contract
|
||||
# ./ctrl/theme.sh parts # what can be added, and the markup that adds it
|
||||
# ./ctrl/theme.sh export [name...] # the contract for a subset, as one doc
|
||||
#
|
||||
# `export` is for handing a vetted LLM what it needs to write an ad-hoc page —
|
||||
# the chosen parts, their markup, and the tokens resolved to literal values, so
|
||||
# the document stands alone. Naming parts is the point: hand over everything and
|
||||
# you get back a page built from Vue components that cannot run standalone.
|
||||
#
|
||||
# ./ctrl/theme.sh export panel split > /tmp/contract.md
|
||||
#
|
||||
# Call the script directly when piping; `make` echoes its recipe to stdout.
|
||||
# For whole-repo context this is the wrong tool — station/tools/distill already
|
||||
# flattens a tree to one budgeted document.
|
||||
#
|
||||
# A page that says `background: var(--bg)` and never gets `--bg` is UNSTYLED,
|
||||
# not merely unbranded — the declaration is invalid at computed-value time. That
|
||||
# is why every page carries a baked default, and why `check` is worth running.
|
||||
#
|
||||
# This exists because bake.py was reachable by no command at all: not from the
|
||||
# Makefile, not from ctrl/, not from build.py. A drift check nobody runs is a
|
||||
# drift check that reports nothing, and the evidence was already on disk —
|
||||
# histgen's page linked /theme.css for months, was missing from the old
|
||||
# hardcoded page list, and so was never baked once.
|
||||
set -e
|
||||
|
||||
# Where the caller stood. A run file is named relative to there, and the cd below
|
||||
# would otherwise make `run ./theme.toml` mean a different file.
|
||||
CALLER_DIR="$PWD"
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
ROOT_DIR="$(dirname "$SCRIPT_DIR")"
|
||||
cd "$ROOT_DIR/soleprint"
|
||||
|
||||
PYTHON="${PYTHON:-python3}"
|
||||
|
||||
case "${1:-bake}" in
|
||||
bake) exec "$PYTHON" common/theme/bake.py ;;
|
||||
check) exec "$PYTHON" common/theme/bake.py --check ;;
|
||||
parts)
|
||||
exec "$PYTHON" common/theme/bake.py --parts
|
||||
;;
|
||||
run)
|
||||
# A run file: every page a project has, its context, and a contract per
|
||||
# page for the LLM. See soleprint/common/theme/theme.example.toml.
|
||||
shift
|
||||
args=() file="" prev=""
|
||||
for a in "$@"; do
|
||||
if [[ -z "$file" && "$a" != -* && "$prev" != "--only" ]]; then
|
||||
file="$(cd "$CALLER_DIR" && realpath -m -- "$a")"
|
||||
args+=("$file")
|
||||
else
|
||||
args+=("$a")
|
||||
fi
|
||||
prev="$a"
|
||||
done
|
||||
exec "$PYTHON" common/theme/bake.py --run "${args[@]}"
|
||||
;;
|
||||
new)
|
||||
shift
|
||||
exec "$PYTHON" common/theme/bake.py --new "$@"
|
||||
;;
|
||||
export)
|
||||
shift
|
||||
exec "$PYTHON" common/theme/bake.py --export "$@"
|
||||
;;
|
||||
*)
|
||||
echo "Unknown: $1" >&2
|
||||
echo "Usage: ./ctrl/theme.sh [new [title]|parts|bake|check|export [name...]|run FILE]" >&2
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
@@ -57,8 +57,7 @@ compose, the same dependency installs as a rig addon of that name:
|
||||
|
||||
```bash
|
||||
cd rig
|
||||
PROFILE=data make cluster up
|
||||
PROFILE=data make addons install
|
||||
PROFILE=data make cluster up # installs the addons too
|
||||
|
||||
kubectl -n data port-forward svc/postgres 5432:5432
|
||||
kubectl -n data port-forward svc/airflow 8080:8080
|
||||
|
||||
18
rig/.gitignore
vendored
18
rig/.gitignore
vendored
@@ -9,11 +9,21 @@ ctrl/.env
|
||||
# generated: the .dot is a build artifact rendered from arch/*.json, never hand-edited.
|
||||
# The .svg IS committed — onboarding material should render in a repo browser.
|
||||
arch/*.dot
|
||||
ctrl/Tiltfile.gen
|
||||
# ctrl/Tiltfile.gen was here for a generator that no longer exists. ctrl/Tiltfile
|
||||
# is now a real, committed file that derives its values when Tilt parses it, so
|
||||
# there is nothing generated to ignore.
|
||||
|
||||
# binaries pulled by `make deps-bundle` for the air-gapped installer image
|
||||
vendor
|
||||
|
||||
# Client rigs are NOT ignored here. A copy is a SIBLING of this directory
|
||||
# (../acme-rig), so a rule in this file cannot see it — the rules live in the
|
||||
# parent repo's .gitignore, anchored at its root, where `*-rig/` matches them.
|
||||
# Overlays that live with no version control of their own, or clones of their own
|
||||
# repos: rig reads them and never tracks them (docs/notes/overlay.md).
|
||||
/local/
|
||||
|
||||
# A profile you activated (cp env.d/<name>.env.example env.d/<name>.env) is this
|
||||
# machine's choice; only the examples are committed.
|
||||
ctrl/env.d/*.env
|
||||
|
||||
# Only the default kit is committed; a kit for a local profile is this machine's.
|
||||
standalone/*
|
||||
!standalone/default/
|
||||
|
||||
114
rig/BOOTSTRAP.md
114
rig/BOOTSTRAP.md
@@ -1,18 +1,9 @@
|
||||
# From a machine with nothing on it to a project you can work in
|
||||
# From a machine with nothing on it to an environment you can work in
|
||||
|
||||
The README says the prerequisite is Docker and nothing else. This is what that
|
||||
actually looks like end to end: a bare Linux box, and a new project running under
|
||||
actually looks like end to end: a bare Linux box, and an overlay running under
|
||||
Tilt at the end of it.
|
||||
|
||||
A copy of this directory is a sibling of it, named after the environment it
|
||||
models (`acme-rig`). Paths below are relative to the parent checkout.
|
||||
|
||||
It spans three repos because the work does. **rig** prepares the machine — the
|
||||
pinned toolchain, the cluster, the port arithmetic. **all** owns the shape a
|
||||
project takes, in `all/projects/templates/conventions.md` and the `broad`
|
||||
scaffold beside it. **ppl** owns everything after local, and is where this
|
||||
document stops.
|
||||
|
||||
Read it once before running anything. Three of the steps below need root and one
|
||||
needs a logout, so knowing about them in advance is cheaper than meeting them
|
||||
halfway through.
|
||||
@@ -97,7 +88,7 @@ build has a Makefile target today. **On a genuinely bare machine, run it by
|
||||
hand:**
|
||||
|
||||
```bash
|
||||
make deps-image # builds rig-deps:deps
|
||||
make deps image # builds rig-deps:deps
|
||||
mkdir -p ~/.local/bin
|
||||
docker run --rm \
|
||||
-v /:/host:ro \
|
||||
@@ -108,7 +99,7 @@ docker run --rm \
|
||||
```
|
||||
|
||||
The image name follows the directory, like everything else here: in `rig`
|
||||
it is `rig-deps`, in a copy called `acme-rig` it is `acme-rig-deps`. The
|
||||
it is `rig-deps`, in a copy called `other-rig` it is `other-rig-deps`. The
|
||||
tag is `deps` (or `full`, below), not `latest`.
|
||||
|
||||
None of the four arguments are guessable, so:
|
||||
@@ -141,7 +132,7 @@ If something else on this machine already provides `kubectl`, the installer says
|
||||
by name rather than shadowing it quietly. `OUT_BIN=$PWD/def/bin` installs
|
||||
somewhere private instead.
|
||||
|
||||
**Two variants worth knowing before you need them.** `make deps-image full` bakes
|
||||
**Two variants worth knowing before you need them.** `make deps image full` bakes
|
||||
every pinned binary into the image at build time (`DEPS_SOURCE=baked`), so
|
||||
`docker save` gives you the entire installer as one file to carry into an
|
||||
air-gapped network. And `DEPS_SOURCE=artifactory` with `DEPS_ARTIFACTORY_URL`
|
||||
@@ -156,18 +147,16 @@ the first-time path.
|
||||
## Prove the machine before blaming the project
|
||||
|
||||
```bash
|
||||
make setup
|
||||
make check
|
||||
make cluster up
|
||||
kubectl get nodes
|
||||
```
|
||||
|
||||
`make setup` re-runs every check as a group. It is idempotent and it deliberately
|
||||
does not abort on the first failure — a setup script that dies at step two hides
|
||||
the fact that steps four and five were also going to fail. Run now, it should be
|
||||
`ok` and `done` all the way down, and that is the point: it is the scoreboard,
|
||||
not the installer.
|
||||
`make check` re-runs every check — host, docker, toolchain, memory, ports — and
|
||||
changes nothing. Run now, it should end with nothing left to do by hand, and that
|
||||
is the point: it is the scoreboard, not the installer.
|
||||
|
||||
`make cluster up` builds the default `minimal` profile — one node, no addons,
|
||||
`make cluster up` builds rig's built-in defaults — one node, no addons,
|
||||
boots fast. You do not need it to develop anything, but you do want to know that
|
||||
kind, the kubeconfig context and the derived port block work *before* a new
|
||||
project has any problems of its own to confuse them with. `make cluster down`
|
||||
@@ -185,70 +174,30 @@ worth reading before rather than after. `make cluster free <names>` stops
|
||||
clusters without deleting them; `docker start` brings them back untouched.
|
||||
|
||||
|
||||
## Scaffold the project
|
||||
## Start an overlay
|
||||
|
||||
The canonical layout is [`all/projects/templates/conventions.md`](../all/projects/templates/conventions.md).
|
||||
Read it — it is short, opinionated, and exists precisely so nobody
|
||||
reverse-engineers a layout from whichever repo they happened to open. What
|
||||
follows is only the mechanical part.
|
||||
What runs lives outside rig, in an overlay — see
|
||||
[`docs/notes/overlay.md`](docs/notes/overlay.md). Start from rig's own:
|
||||
|
||||
```bash
|
||||
SLUG=<slug> # short, lowercase, no separators
|
||||
cp -r ~/wdir/semester/all/projects/templates/broad ~/wdir/semester/"$SLUG"
|
||||
cd ~/wdir/semester/"$SLUG"
|
||||
grep -rl '<slug>' ctrl | xargs sed -i "s/<slug>/$SLUG/g"
|
||||
cp ctrl/k8s/.env.example ctrl/k8s/.env
|
||||
git init && git add -A && git commit -m "scaffold $SLUG from broad"
|
||||
cp -r examples/starter local/myenv # local/ is gitignored by rig
|
||||
echo 'OVERLAY=local/myenv' >> ctrl/.env
|
||||
make check # shows the overlay, its cluster and ports
|
||||
```
|
||||
|
||||
`<slug>` is the only placeholder and it lives only under `ctrl/` — cluster name,
|
||||
namespace, ConfigMap name, and the `NAME=` in `kind-up.sh` / `kind-down.sh`. One
|
||||
sed does all of it.
|
||||
|
||||
The slug is the folder name, lowercase and short — `mpr`, `unt`, `nvi`. The
|
||||
cluster takes that name and the context becomes `kind-<slug>`, derived by the
|
||||
scaffold's Makefile from the directory, so there is nothing to edit for either.
|
||||
|
||||
**Pick the Tilt port deliberately.** `ctrl/k8s/.env.example` ships a value that
|
||||
is already in use, so copying it unchanged puts two projects on one port:
|
||||
The cluster takes the overlay's folder name and the context becomes
|
||||
`kind-<name>`, so there is nothing to edit for either. Replace the two example
|
||||
components under `k8s/base/`, and add your images and resources to the
|
||||
overlay's `Tiltfile`. Check the manifests before `kind` spends minutes on anything
|
||||
— this renders the whole tree without a cluster and catches a broken patch
|
||||
immediately:
|
||||
|
||||
```bash
|
||||
grep -h '^TILT_PORT=' ~/wdir/semester/*/ctrl/k8s/.env 2>/dev/null | sort
|
||||
kubectl kustomize local/myenv/k8s/overlays/dev
|
||||
```
|
||||
|
||||
Choose a free one in `10300–10399` — the range ALL reserves in
|
||||
`projects/index.json` under `policy` — avoiding `10350`, which is Tilt's own
|
||||
default. Currently taken: `nvi` 10330, `unt` 10340, `mpr` 10360, `mlv` 10370,
|
||||
`eth` 10380, `lng` 10390. This is the Tilt *web UI* port, not a service port;
|
||||
each project owns its own service ports separately. The scaffold ships it blank
|
||||
on purpose, so there is nothing to collide with until you choose.
|
||||
|
||||
The scaffold's `ctrl/k8s/` is the same shape as every other project here, and it
|
||||
builds as shipped:
|
||||
|
||||
```
|
||||
kind-config.yaml one node; gateway NodePort 30080 -> hostPort 8080
|
||||
base/ namespace, configmap, app (Deployment + Service)
|
||||
overlays/dev/ promotes the app Service to NodePort 30080
|
||||
```
|
||||
|
||||
Check it before `kind` spends minutes on anything — this renders the whole tree
|
||||
without a cluster and catches a broken patch immediately:
|
||||
|
||||
```bash
|
||||
kubectl kustomize ctrl/k8s/overlays/dev
|
||||
```
|
||||
|
||||
The workload is an nginx placeholder so a fresh copy reaches something that
|
||||
answers; replace it. Keep `30080` in step between the overlay patch and
|
||||
`kind-config.yaml`'s `containerPort` — the hostPort is this project's to pick.
|
||||
Reachability is a plain kind port mapping: no ingress controller and no MetalLB.
|
||||
Caddy maps `<slug>.local.ar` onto the host port (`~/wdir/semester/ppl/local/Caddyfile`),
|
||||
with `*.local.ar` resolving to 127.0.0.1 through dnsmasq. That is the whole chain.
|
||||
|
||||
**The one file the scaffold still does not ship is `ctrl/Tiltfile`** — `make
|
||||
tilt-up` runs `cd ctrl && tilt up`, and there is nothing to run until you write
|
||||
one. Copy it from a live project; `unt` and `nvi` are closest to the plain shape.
|
||||
If the overlay is to be versioned, make `local/myenv` a repository of its own (rig
|
||||
never tracks it), or keep it anywhere else and name it by path.
|
||||
|
||||
|
||||
## Run it
|
||||
@@ -258,20 +207,9 @@ make kind-up # idempotent create, then selects the context
|
||||
make tilt-up # context + your assigned port
|
||||
```
|
||||
|
||||
`tilt-up` passes `--context kind-<slug>` every time, which is the point of going
|
||||
`tilt-up` passes `--context kind-<name>` every time, which is the point of going
|
||||
through `make` at all: tilt cannot deploy into whichever cluster you last looked
|
||||
at.
|
||||
|
||||
`make tilt-down` and `make kind-down` close the loop, and `make kind-reset` is
|
||||
delete-and-recreate for when a cluster wedges.
|
||||
|
||||
|
||||
## Register it
|
||||
|
||||
The project exists; now it is findable. Add an entry to
|
||||
`~/wdir/semester/all/projects/index.json` and write its `projects/<slug>.md` beside the
|
||||
others. Structured fields in the index, prose in the markdown.
|
||||
|
||||
Putting it on the CI server and deploying it is `ppl`'s half, and it starts at
|
||||
`~/wdir/semester/ppl/ctrl/init-repo.sh` — gitea remote, then Woodpecker. That is a
|
||||
different document.
|
||||
|
||||
136
rig/Makefile
136
rig/Makefile
@@ -1,25 +1,18 @@
|
||||
# Thin control Makefile — one target per ctrl/ script, and the subcommand is an
|
||||
# argument rather than a second target: `make cluster down`, not `make cluster-down`.
|
||||
#
|
||||
# The logic lives in the scripts, never here. Each target maps to exactly one
|
||||
# bash file, and that file holds the variants:
|
||||
#
|
||||
# make cluster up -> ctrl/cluster.sh up
|
||||
# make newbox destroy -> ctrl/newbox.sh destroy
|
||||
#
|
||||
# Config layers, weakest first: ctrl/versions.env (pinned toolchain) <
|
||||
# ctrl/env.d/<profile>.env (cluster shape) < ctrl/.env (local, gitignored) <
|
||||
# the environment. So `make cluster up PROFILE=client` beats everything.
|
||||
#
|
||||
# Start with: make setup (then: make cluster up && make docs)
|
||||
|
||||
# Identity follows the FOLDER NAME, so this directory can be copied elsewhere,
|
||||
# renamed, and run as a separate environment with no edits. ctrl/.env overrides
|
||||
# it when you want a name that differs from the directory.
|
||||
# Thin control Makefile: the subcommand is an argument (`make cluster down`); logic lives in ctrl/ scripts.
|
||||
# make check | deps | cluster up | tilt | docs (`make help` lists all)
|
||||
# Start with: make check && make deps && make cluster up
|
||||
# Notes: docs/notes/Makefile.md
|
||||
|
||||
# Identity and ports, asked once of ctrl/ports.sh, read positionally (selftest pins the order):
|
||||
# CLUSTER KUBECONTEXT HTTP HTTPS TILT REGISTRY MANIFESTS_DIR OVERLAY_DIR
|
||||
# OVERLAY/CLUSTER given as make arguments are handed over explicitly: before make 4.4,
|
||||
# $(shell) does not see them, and the context would follow the wrong environment.
|
||||
FACTS := $(shell $(if $(OVERLAY),OVERLAY='$(OVERLAY)') $(if $(CLUSTER),CLUSTER='$(CLUSTER)') bash ctrl/ports.sh active 2>/dev/null)
|
||||
SLUG := $(shell echo '$(notdir $(CURDIR))' | tr '[:upper:]' '[:lower:]' | tr -c 'a-z0-9-' '-' | sed 's/^-*//; s/-*$$//')
|
||||
CLUSTER := $(or $(shell sed -n 's/^CLUSTER=//p' ctrl/.env 2>/dev/null),$(SLUG))
|
||||
KCTX := --context kind-$(CLUSTER)
|
||||
TILT_PORT := $(shell sed -n 's/^TILT_PORT=//p' ctrl/.env 2>/dev/null)
|
||||
# Fall back to the folder name, not empty, if ports.sh fails on a broken config.
|
||||
CLUSTER := $(or $(word 1,$(FACTS)),$(SLUG))
|
||||
KCTX := --context $(or $(word 2,$(FACTS)),kind-$(SLUG))
|
||||
TILT_PORT := $(word 5,$(FACTS))
|
||||
DEPSIMG := $(SLUG)-deps
|
||||
|
||||
# Words after the target become the script's subcommand. Make would otherwise
|
||||
@@ -27,89 +20,68 @@ DEPSIMG := $(SLUG)-deps
|
||||
ARGS := $(wordlist 2,$(words $(MAKECMDGOALS)),$(MAKECMDGOALS))
|
||||
ifneq ($(ARGS),)
|
||||
$(eval $(ARGS):;@:)
|
||||
# ...and as PHONY, because some of those words name real directories. `cfg`,
|
||||
# `ctrl`, `docs`, `gen` and `init` all exist at this level, and make considers a
|
||||
# target that is an existing directory already built — so `make build ctrl` ran
|
||||
# the build and then printed "make: 'ctrl' is up to date". The empty rule above
|
||||
# is not enough on its own; only .PHONY stops make consulting the filesystem.
|
||||
# ...and as PHONY, because some of those words name real directories (ctrl, docs, ...).
|
||||
.PHONY: $(ARGS)
|
||||
endif
|
||||
|
||||
.PHONY: help setup check mem deps deps-image pins cluster registry addons ports \
|
||||
newbox dockerhost docs tilt \
|
||||
.PHONY: help check deps cluster tilt docs selftest standalone \
|
||||
kind-up kind-down kind-reset tilt-up tilt-down
|
||||
|
||||
help: ## list targets
|
||||
@grep -hE '^[a-z][a-z-]*:.*?##' $(MAKEFILE_LIST) | sed 's/:.*##/\t/' | expand -t16
|
||||
|
||||
# ── setup ──────────────────────────────────────────────────────────────────
|
||||
# ── this machine ───────────────────────────────────────────────────────────
|
||||
|
||||
setup: ## prepare this machine [core] [--share-docker] [--cluster]
|
||||
bash ctrl/setup.sh $(ARGS)
|
||||
# Everything that looks and never changes anything: host, docker, toolchain,
|
||||
# config, memory, ports, registry, addons. `check mem` goes deeper on memory —
|
||||
# how far it really climbs, and the WSL .wslconfig backup/restore.
|
||||
check: ## is this machine ready? [all] [mem [status|push|all|backup|restore]]
|
||||
bash ctrl/check.sh $(ARGS)
|
||||
|
||||
check: ## is this machine ready? reports, never fixes
|
||||
bash ctrl/check.sh
|
||||
|
||||
mem: ## memory, and any cap holding it [status|backup|restore]
|
||||
bash ctrl/mem.sh $(or $(ARGS),status)
|
||||
|
||||
deps: ## install the toolchain [core|dev] (default dev)
|
||||
bash ctrl/deps.sh install $(or $(ARGS),dev)
|
||||
|
||||
pins: ## standalone/rigdeps.sh still installs what rig pins?
|
||||
bash ctrl/pins.sh
|
||||
|
||||
deps-image: ## build the installer image [full]
|
||||
# `deps image` is for a machine with nothing but Docker: the installer runs from
|
||||
# the image instead — see BOOTSTRAP.md. `full` bakes every binary in.
|
||||
deps: ## install the toolchain [core|dev] [image [full]]
|
||||
ifeq ($(word 1,$(ARGS)),image)
|
||||
docker build -f ctrl/Dockerfile.deps \
|
||||
--target $(if $(filter full,$(ARGS)),deps-full,deps) \
|
||||
-t $(DEPSIMG):$(if $(filter full,$(ARGS)),full,deps) .
|
||||
else
|
||||
bash ctrl/deps.sh install $(or $(ARGS),dev)
|
||||
endif
|
||||
|
||||
# ── cluster ────────────────────────────────────────────────────────────────
|
||||
# ── the cluster ────────────────────────────────────────────────────────────
|
||||
|
||||
# up also starts the registry and installs the profile's addons, and the ports
|
||||
# derive from the folder name — there is nothing else to run first.
|
||||
cluster: ## this env + the machine [up|down|reset|list|free]
|
||||
bash ctrl/cluster.sh $(or $(ARGS),up)
|
||||
|
||||
registry: ## registry wiring [up|down|status] (default status)
|
||||
bash ctrl/registry.sh $(or $(ARGS),status)
|
||||
|
||||
addons: ## profile addons [install|list] (default list)
|
||||
bash ctrl/addons.sh $(or $(ARGS),list)
|
||||
|
||||
ports: ## this environment's port block [show|persist]
|
||||
bash ctrl/ports.sh $(or $(ARGS),show)
|
||||
|
||||
# ── host ───────────────────────────────────────────────────────────────────
|
||||
|
||||
newbox: ## throwaway environment [create|status|shell|destroy]
|
||||
bash ctrl/newbox.sh $(or $(ARGS),status)
|
||||
|
||||
dockerhost: ## share Docker between distros [status|share|unshare]
|
||||
$(if $(filter share unshare,$(ARGS)),sudo ,)bash ctrl/dockerhost.sh $(or $(ARGS),status)
|
||||
|
||||
# ── docs + dev loop ────────────────────────────────────────────────────────
|
||||
# ── dev loop + docs ────────────────────────────────────────────────────────
|
||||
|
||||
docs: ## documentation [serve|graphs] (default serve)
|
||||
bash ctrl/docs.sh $(or $(ARGS),serve)
|
||||
|
||||
# --port is only passed when TILT_PORT is actually set. It comes from ctrl/.env,
|
||||
# which does NOT carry it by default — ports are derived at runtime in
|
||||
# lib/config.sh unless `make ports persist` has written them. Without the guard
|
||||
# tilt receives a bare `--port` with no value and fails on the flag rather than
|
||||
# on anything real. `make ports show` prints the derived block.
|
||||
# --port only when TILT_PORT resolved; the Tiltfile asks ports.sh for the rest itself.
|
||||
tilt: ## dev loop [up|down] (default up)
|
||||
cd ctrl && tilt $(or $(ARGS),up) $(KCTX) $(if $(filter down,$(ARGS)),,$(if $(TILT_PORT),--port $(TILT_PORT)))
|
||||
|
||||
# ── maintaining rig ────────────────────────────────────────────────────────
|
||||
|
||||
# The counterpart to check: that one asks about the MACHINE and never fails,
|
||||
# this one asks about RIG and exits 1. Checks are written as the decisions they
|
||||
# defend, so a failure names what is being undone.
|
||||
selftest: ## does rig still do what it says? exits 1 if not
|
||||
bash ctrl/selftest.sh
|
||||
|
||||
# The one-file versions of rig's tools, one folder per profile, for machines the
|
||||
# full rig is not going to. Generated from rig as it is, never edited by hand;
|
||||
# `check` is what selftest runs to catch a kit left behind by a change to rig.
|
||||
standalone: ## single-file kits [write|check|export DIR] (default write)
|
||||
bash ctrl/standalone.sh $(or $(ARGS),write)
|
||||
|
||||
# ── the shape every other project uses ─────────────────────────────────────
|
||||
# Aliases, not a second implementation: each one calls the same script the
|
||||
# canonical target does.
|
||||
#
|
||||
# The header above argues for `make cluster down` over `make cluster-down`, and
|
||||
# that still holds *within* this file. But rig is one repo among several on the
|
||||
# same machine, and every other one answers to kind-up / tilt-up. Muscle memory
|
||||
# spanning six projects beats internal tidiness in one, so both spellings work.
|
||||
#
|
||||
# `cluster list` and `cluster free` have no hyphenated twin on purpose — they
|
||||
# are rig's own, with nothing to be consistent with.
|
||||
# Aliases matching other projects' kind-up / tilt-up; each calls the same script.
|
||||
# Nothing else reads these names: rename, delete or add freely (and update .PHONY).
|
||||
|
||||
kind-up: ## alias for `cluster up`
|
||||
bash ctrl/cluster.sh up
|
||||
@@ -120,10 +92,10 @@ kind-down: ## alias for `cluster down`
|
||||
kind-reset: ## alias for `cluster reset`
|
||||
bash ctrl/cluster.sh reset
|
||||
|
||||
# These two match the other projects' spelling, but rig has no Tiltfile — there
|
||||
# is nothing to run yet, and they fail the same way `make tilt` does.
|
||||
tilt-up: ## alias for `tilt up` (rig has no Tiltfile yet)
|
||||
# These two match the other projects' spelling. rig ships ctrl/Tiltfile, so they
|
||||
# run — it deploys the examples in k8s/base until you replace them.
|
||||
tilt-up: ## alias for `tilt up`
|
||||
cd ctrl && tilt up $(KCTX) $(if $(TILT_PORT),--port $(TILT_PORT))
|
||||
|
||||
tilt-down: ## alias for `tilt down` (rig has no Tiltfile yet)
|
||||
tilt-down: ## alias for `tilt down`
|
||||
cd ctrl && tilt down $(KCTX)
|
||||
|
||||
148
rig/README.md
148
rig/README.md
@@ -58,40 +58,57 @@ make docs # serves on localhost, prints the URL
|
||||
They run before anything is installed, which matters because they are the
|
||||
instructions for everything else. No cluster and no toolchain required.
|
||||
|
||||
Why the code is the way it is — the reasoning, measurements and gotchas — lives
|
||||
in [`docs/notes/`](docs/notes/), one file per script, so the code keeps short comments.
|
||||
|
||||
## Then
|
||||
|
||||
```bash
|
||||
make check # report host and config problems; changes nothing
|
||||
make check # is this machine ready? short; `make check all` for every detail
|
||||
make deps # install the toolchain (add `core` on a managed machine)
|
||||
make cluster up # build the cluster for the active profile
|
||||
make cluster up # cluster + registry + the profile's addons
|
||||
```
|
||||
|
||||
`make cluster up` also starts this environment's local registry and wires it
|
||||
into the node, so an image built locally is pullable by the cluster without
|
||||
going near docker.io:
|
||||
That is the whole setup. `make cluster up` also starts this environment's local
|
||||
registry and wires it into the node, so an image built locally is pullable by the
|
||||
cluster without going near docker.io. `make check` shows its port, among
|
||||
everything else:
|
||||
|
||||
```bash
|
||||
make registry status # prints: endpoint localhost:<port>
|
||||
make check # ... registry localhost:<port> (running)
|
||||
docker build -t localhost:<port>/app:1 .
|
||||
docker push localhost:<port>/app:1
|
||||
kubectl --context kind-$(basename $PWD) run app --image=localhost:<port>/app:1
|
||||
```
|
||||
|
||||
The port block is derived from the directory name, so two copies of rig never
|
||||
collide:
|
||||
collide — nothing to configure. `make check all` lists it; `bash ctrl/ports.sh persist`
|
||||
pins it into `ctrl/.env` if you want it fixed:
|
||||
|
||||
```bash
|
||||
make ports show # HTTP / HTTPS / TILT / REGISTRY
|
||||
make cluster list # every cluster on this machine, with memory
|
||||
make cluster free # stop the others if memory is tight
|
||||
make cluster down # remove this cluster and its registry
|
||||
```
|
||||
|
||||
**`make tilt` has nothing to run yet.** The target and its `tilt-up` / `tilt-down`
|
||||
aliases exist so rig answers to the same spelling as every other project here,
|
||||
but rig ships no `Tiltfile` — it builds the estate, it is not itself a service
|
||||
with a dev loop. Add a `ctrl/Tiltfile` and the target works; until then it fails
|
||||
on the missing file, not on anything rig did.
|
||||
**The verbs are yours to change.** `cluster` is the script — `ctrl/cluster.sh` —
|
||||
and every spelling above is a `Makefile` target that calls it. `make kind-up` is
|
||||
an alias for `make cluster up`, kept because the other projects on this machine
|
||||
answer to that spelling and muscle memory spans repos rather than stopping at
|
||||
one. Nothing outside the `Makefile` reads these names, so rename them, drop the
|
||||
ones you never type, or add whatever your own projects already say: each alias
|
||||
is two lines at the bottom of the file, calling the same script the canonical
|
||||
target does.
|
||||
|
||||
**`make tilt` works on a fresh copy, unedited.** rig's `ctrl/Tiltfile` does
|
||||
rig's part — identity, context guard, registry, the manifests — and then includes
|
||||
the overlay's own Tiltfile. With no overlay named that is `examples/starter`, so
|
||||
the dev loop comes up with its two examples running and nothing to configure.
|
||||
|
||||
It hardcodes nothing. It asks `ctrl/ports.sh active` for this environment's
|
||||
cluster name, kube context, ports and paths — the same values every other rig
|
||||
script resolves through `ctrl/lib/config.sh` — so a copied and renamed rig, or a
|
||||
moved overlay, deploys into its own cluster with no edits.
|
||||
|
||||
`make help` lists every target.
|
||||
|
||||
@@ -100,85 +117,82 @@ nothing to download with — see [BOOTSTRAP.md](BOOTSTRAP.md), which runs the
|
||||
toolchain through the installer container and carries on to scaffolding and running
|
||||
a new project.
|
||||
|
||||
## One directory is one environment
|
||||
## What runs is an overlay
|
||||
|
||||
rig is the machine: toolchain, cluster, registry, port block, the dev loop's
|
||||
plumbing. What runs on it is an **overlay** — one folder, outside rig's version
|
||||
control, holding a use case: its settings (`rig.env`), its manifests
|
||||
(`k8s/overlays/dev`), its images and Tiltfile, its addons, its kind config if it
|
||||
needs its own. rig reads it and never writes into it. See
|
||||
[`docs/notes/overlay.md`](docs/notes/overlay.md).
|
||||
|
||||
```bash
|
||||
cp -r examples/starter local/myenv # local/ is gitignored
|
||||
OVERLAY=local/myenv make cluster up # or OVERLAY=local/myenv in ctrl/.env
|
||||
OVERLAY=local/myenv make tilt
|
||||
```
|
||||
|
||||
An overlay can also be a repo of its own, anywhere, or a project folder that
|
||||
carries rig at `<project>/rig/` with a three-line forwarding Makefile.
|
||||
|
||||
Copy this directory, rename it, run it. Cluster name, kubectl context, image
|
||||
tags and the host port block all derive from the directory name, so copies never
|
||||
collide and neither one's teardown can touch the other.
|
||||
## One environment per folder
|
||||
|
||||
A copy of this directory is a **sibling** of it, named after the environment it
|
||||
models (`acme-rig`). That is why the ignore rules for copies sit in the *parent*
|
||||
repo's `.gitignore` rather than here: a rule in this directory cannot see a
|
||||
directory beside it.
|
||||
Cluster name, kubectl context, image tags and the host port block all derive
|
||||
from a folder name — the overlay's when one is named, else rig's own — so copies
|
||||
never collide and neither one's teardown can touch the other. Two overlays run
|
||||
side by side from one rig; two copies of rig do too.
|
||||
|
||||
## Profiles
|
||||
|
||||
A profile is the shape of the cluster: how many nodes, which addons, whether the
|
||||
apiserver audits. They live in `ctrl/env.d/`, and the active one is `PROFILE`.
|
||||
**rig needs no profile.** With none named it runs on built-in defaults: one node,
|
||||
no addons, a local registry, the newest Kubernetes version it pins. A profile is
|
||||
an optional file in `ctrl/env.d/`, named by `PROFILE`, that says how this machine
|
||||
reaches the world. rig ships two as **examples**; copy one to use it (the copy is
|
||||
gitignored):
|
||||
|
||||
| Profile | For |
|
||||
| example | what it changes |
|
||||
| --- | --- |
|
||||
| `minimal` | the default. One node, no addons, boots fast. |
|
||||
| `client` | the regulated-estate shape — multi-node, audit on, registry mirror. |
|
||||
| `offline` | air-gapped: everything from a preloaded local registry. |
|
||||
| `data` | the cabinets an environment asks for. |
|
||||
| `mirror.env.example` | images through a pull-through cache of an internal registry |
|
||||
| `offline.env.example` | air-gapped: everything from a preloaded local registry |
|
||||
|
||||
```bash
|
||||
PROFILE=data make cluster up
|
||||
PROFILE=data make addons install
|
||||
make addons # what the active profile wants, and what exists
|
||||
cp ctrl/env.d/mirror.env.example ctrl/env.d/mirror.env
|
||||
PROFILE=mirror make cluster up
|
||||
```
|
||||
|
||||
A profile names a **cluster shape** — a file in `ctrl/k8s/` — rather than
|
||||
restating node count and audit as variables:
|
||||
|
||||
| shape | nodes | audit | used by |
|
||||
| --- | --- | --- | --- |
|
||||
| `kind-config.yaml.tpl` | 1 | off | `minimal`, `data` |
|
||||
| `kind-config.audit.yaml.tpl` | 1 | on | `offline` |
|
||||
| `kind-config.client.yaml.tpl` | 3 | on | `client` |
|
||||
The layers, weakest first: built-in defaults < `ctrl/versions.env` < the profile
|
||||
< the overlay's `rig.env` < `ctrl/.env` < your command line.
|
||||
|
||||
Both numbers are read back out of the chosen file, so the YAML is the only place
|
||||
that decides and there is nothing to drift. The layout under `ctrl/k8s/` is the
|
||||
same as every other project here — a kind config, a kustomize `base/`, an
|
||||
`overlays/dev/` — see [`ctrl/k8s/README.md`](ctrl/k8s/README.md).
|
||||
The **cluster itself** is one file: the overlay's `kind-config.yaml.tpl` if it has
|
||||
one, else rig's `ctrl/k8s/kind-config.yaml.tpl` (one node). To change it — more
|
||||
nodes, other port mappings, mounts — edit it and `make cluster reset`. The node
|
||||
count is read back out of it, so there is nothing to drift.
|
||||
|
||||
## Addons
|
||||
|
||||
Each addon is its own idempotent script in `ctrl/addons/`, and a profile names
|
||||
the ones it wants in `ADDONS`. Adding one is adding a file — there is no
|
||||
dispatcher to edit.
|
||||
Each addon is its own idempotent script, and `ADDONS` names the ones to install,
|
||||
in order. Adding one is adding a file — there is no dispatcher to edit. An
|
||||
overlay's `addons/<name>.sh` is found before rig's own.
|
||||
|
||||
**There is no ingress controller, deliberately.** They pin a narrow window of
|
||||
Kubernetes versions, so depending on one would constrain which k8s a rig can be
|
||||
built with — and running a trailing-edge control plane to model a legacy estate
|
||||
is the whole point. Services are reached through MetalLB and
|
||||
`type: LoadBalancer`, which carries no such constraint and is also what a real
|
||||
cluster does.
|
||||
built with — and running a trailing-edge control plane is often the point.
|
||||
Services are reached through MetalLB and `type: LoadBalancer`, which carries no
|
||||
such constraint and is also what a real cluster does.
|
||||
|
||||
rig's own addons make the cluster work:
|
||||
|
||||
| Addon | Does |
|
||||
| --- | --- |
|
||||
| `metallb` | gives `type: LoadBalancer` an address it can actually reach |
|
||||
| `cert-manager` | a local CA, so TLS works offline |
|
||||
| `metrics-server` | makes `kubectl top` work on kind |
|
||||
| `postgres` | database, in the `data` namespace |
|
||||
| `redis` | cache and broker |
|
||||
| `airflow` | scheduled pipelines; needs postgres and redis |
|
||||
|
||||
The last three are **cabinets**: a public service dropped in as-is, the upstream
|
||||
image unmodified, reachable at a known address. A cabinet is declared once and
|
||||
installs on either target — a `service.yml` composes it for a laptop, and these
|
||||
install the same one here. The names match on purpose: each cabinet carries a
|
||||
`rig_addon` field pointing at `ctrl/addons/<name>.sh`.
|
||||
|
||||
Plain manifests rather than helm charts, like every other addon: a chart repo is
|
||||
a network dependency, and the `offline` profile exists precisely so there is a
|
||||
path with none. Images are pinned in `ctrl/versions.env` and can be preloaded.
|
||||
|
||||
Passwords are generated on first install and kept across re-runs, so re-running
|
||||
an addon never rotates a credential out from under something already connected:
|
||||
What a workload needs — a database, a cache, a scheduler — belongs to its
|
||||
overlay. [`examples/data`](examples/data/) carries postgres, redis and airflow as
|
||||
plain manifests (no helm: a chart repo is a network dependency), with passwords
|
||||
generated on first install and kept across re-runs:
|
||||
|
||||
```bash
|
||||
kubectl -n data get secret postgres -o jsonpath='{.data.POSTGRES_PASSWORD}' | base64 -d
|
||||
kubectl -n data port-forward svc/airflow 8080:8080
|
||||
OVERLAY=examples/data make cluster up
|
||||
```
|
||||
|
||||
68
rig/STALE.md
Normal file
68
rig/STALE.md
Normal file
@@ -0,0 +1,68 @@
|
||||
# rig — withdrawn assumptions
|
||||
|
||||
**Everything in this file is no longer true.**
|
||||
|
||||
It exists so the live docs stay short and a withdrawn assumption cannot quietly
|
||||
return: each entry carries a **check**, and `ctrl/selftest.sh` runs every one of
|
||||
them under "withdrawn stays withdrawn". A retraction that is only prose is one
|
||||
nobody re-reads.
|
||||
|
||||
- **Do not restate these** in a plan, a README or a comment; point here (`✖ S2`).
|
||||
- **Ids are stable.**
|
||||
- **Only withdrawn things belong here.** A warning that is still actionable is a
|
||||
live rule and stays where it is.
|
||||
|
||||
---
|
||||
|
||||
**✖ S1 — "rig supplies `ctrl/Tiltfile` but does not own it: replace the examples and
|
||||
add your images in its marked sections."** *(ctrl/Tiltfile, README, 2026-09-13)*
|
||||
Withdrawn 2026-09-22. Every project that used rig edited rig's own file, so no
|
||||
update to rig could land without merging those edits by hand. rig's `ctrl/Tiltfile`
|
||||
is now rig's: identity, context guard, registry, manifests, namespaces. The
|
||||
workload's half is the overlay's own `Tiltfile`, which rig's includes.
|
||||
**Check:** `ctrl/Tiltfile` includes the overlay's Tiltfile and has no "Images" section
|
||||
of its own.
|
||||
|
||||
**✖ S2 — "A second environment is a copy of rig renamed after it (`../<name>-rig`), and
|
||||
its use case is written into the copy."** *(README, BOOTSTRAP, .gitignore, 2026-08)*
|
||||
Withdrawn 2026-09-22. A use case is an overlay, kept outside rig's version control
|
||||
(`docs/notes/overlay.md`); rig itself is replaced as a whole. Copying rig still gives
|
||||
a separate environment, but carries no use case of its own.
|
||||
**Check:** rig's `.gitignore` ignores `/local/`; no `acme` example name remains in rig.
|
||||
|
||||
**✖ S3 — "rig's addons include the services a workload needs (a database, a cache, a
|
||||
scheduler), described in the host project's vocabulary."** *(ctrl/addons/, README,
|
||||
versions.env, 2026-08)* Withdrawn 2026-09-22. Which services a workload needs is
|
||||
not rig's business. `ctrl/addons/` holds what makes a cluster work; workload addons
|
||||
live with overlays, and `examples/data` carries them as an example.
|
||||
**Check:** `ctrl/addons/` holds exactly cert-manager, metallb and metrics-server;
|
||||
`versions.env` pins no workload image.
|
||||
|
||||
**✖ S4 — "The dev loop's namespace is named after the cluster."** *(ctrl/Tiltfile,
|
||||
2026-09-13)* Withdrawn 2026-09-22. The Tiltfile created and grouped
|
||||
`<CLUSTER>:namespace` while the example manifests declared `rig`, so every copy not
|
||||
named `rig` stopped at load: `No object identified by the fragment
|
||||
"acme-rig:namespace"`. The first real use worked around it with a Namespace named
|
||||
`to_be_replaced` that its overlay renamed. The Tiltfile now creates the namespaces
|
||||
the manifests use and groups the ones they declare.
|
||||
**Check:** `ctrl/Tiltfile` does not build a namespace name from `CLUSTER`.
|
||||
|
||||
**✖ S5 — "`MANIFESTS_DIR` defaults to `ctrl/k8s/overlays/dev`, rig's own examples."**
|
||||
*(lib/config.sh, .env.example, 2026-09-13)* Withdrawn 2026-09-22. The default is the
|
||||
overlay's `k8s/overlays/dev`; rig's examples are `examples/starter`. The old value,
|
||||
still pinned by older `.env` files, is ignored while that folder does not exist, and
|
||||
`make check` says to delete it.
|
||||
**Check:** `ctrl/k8s/overlays` does not exist; `.env.example` does not set
|
||||
`MANIFESTS_DIR`.
|
||||
|
||||
**✖ S6 — "The example profiles are `client`, `data` and `offline`."** *(ctrl/env.d/,
|
||||
2026-09-17)* Withdrawn 2026-09-22. "client" names the customer, not a registry
|
||||
mode: the example is `mirror`. `data` was a workload, not a way of reaching the
|
||||
world: it is the `examples/data` overlay.
|
||||
**Check:** `ctrl/env.d/` holds no `client` or `data` example.
|
||||
|
||||
**✖ S7 — "BOOTSTRAP spans three repos: rig prepares the machine, the house repos own the
|
||||
project's shape and everything after local."** *(BOOTSTRAP.md, 2026-08)* Withdrawn
|
||||
2026-09-22. rig's docs describe rig. The house scaffold and registration sections
|
||||
moved out of rig; BOOTSTRAP now ends with starting an overlay.
|
||||
**Check:** no house path or host name (`semester`, `local.ar`) in rig.
|
||||
@@ -1,49 +1,42 @@
|
||||
# Machine-local config. Copy to ctrl/.env (gitignored) and edit.
|
||||
# Cluster SHAPE lives in ctrl/env.d/<profile>.env — not here.
|
||||
# The architecture MODEL lives in arch/<name>.json — not here either.
|
||||
# Cluster SHAPE: an optional profile in ctrl/env.d/. Architecture MODEL: arch/<name>.json.
|
||||
# Notes: docs/notes/env.md
|
||||
|
||||
# Which profile in ctrl/env.d/ to build. minimal | client | offline
|
||||
PROFILE=minimal
|
||||
# A profile in ctrl/env.d/ to build. Empty means rig's built-in defaults, which
|
||||
# need no profile at all. Copy an env.d/*.env.example to <name>.env to add one.
|
||||
PROFILE=
|
||||
|
||||
# The overlay: one folder, outside rig's version control, holding what runs —
|
||||
# its settings (rig.env), manifests, addons, Tiltfile. Relative to rig's folder,
|
||||
# or absolute. Unset: rig's own examples/starter. See docs/notes/overlay.md.
|
||||
# OVERLAY=local/<name>
|
||||
|
||||
# Cluster name; the kubectl context becomes kind-<CLUSTER>.
|
||||
# LEAVE THIS UNSET unless you need a name that differs from the directory —
|
||||
# it defaults to this folder's name, which is what makes the folder copyable:
|
||||
# copy it, rename it, and you get a separate environment with no edits.
|
||||
# LEAVE UNSET: it defaults to this folder's name, which keeps the folder copyable.
|
||||
# CLUSTER=
|
||||
|
||||
# Host ports. LEAVE UNSET — they derive from the directory name so several
|
||||
# environments coexist without negotiating (see ctrl/ports.sh). `make ports`
|
||||
# shows this environment's block; `make ports persist` writes it here so it stops
|
||||
# being derived and becomes fixed. Set a value only to override.
|
||||
# Host ports. LEAVE UNSET — derived from the directory name (see ctrl/ports.sh).
|
||||
# `bash ctrl/ports.sh persist` pins them here; set a value only to override.
|
||||
# HTTP_PORT=
|
||||
# HTTPS_PORT=
|
||||
# TILT_PORT=
|
||||
# REGISTRY_PORT=
|
||||
|
||||
# Where the application manifests live. The real ones are expected to be
|
||||
# versioned separately from this installer — they change on a different cadence,
|
||||
# by different people. Repoint this at their repo and rig stops owning them:
|
||||
# Where the manifests live. Leave unset: the overlay's k8s/overlays/dev.
|
||||
# MANIFESTS_DIR=../platform-manifests/overlays/dev
|
||||
MANIFESTS_DIR=ctrl/k8s/overlays/dev
|
||||
|
||||
# Where the installer fetches the pinned binaries from.
|
||||
# upstream GitHub releases / dl.k8s.io (needs internet)
|
||||
# artifactory a generic repo — what a locked-down client usually allows
|
||||
# baked already inside the installer image; no network at all
|
||||
# Where the installer fetches the pinned binaries from:
|
||||
# upstream (needs internet) | artifactory (generic repo) | baked (in the image)
|
||||
DEPS_SOURCE=upstream
|
||||
DEPS_ARTIFACTORY_URL=
|
||||
|
||||
# --- Registry -------------------------------------------------------------
|
||||
# Mode comes from the profile (REGISTRY_MODE). These are the secrets it needs.
|
||||
# Required for mirror/remote:
|
||||
# Mode comes from the profile (REGISTRY_MODE). Secrets required for mirror/remote:
|
||||
REGISTRY_REMOTE_URL=
|
||||
REGISTRY_USER=
|
||||
REGISTRY_PASSWORD=
|
||||
|
||||
# Corporate root CA, if Artifactory is fronted by an internal CA (it usually is).
|
||||
# Trust has to reach THREE places and nothing does it for you: the host docker
|
||||
# daemon, every kind node's containerd, and any in-cluster client. registry.sh
|
||||
# handles the first two; check.sh reports when it's configured but not trusted.
|
||||
# Corporate root CA, if Artifactory is fronted by an internal CA.
|
||||
# Symptom when missing: x509: certificate signed by unknown authority
|
||||
REGISTRY_CA_FILE=
|
||||
|
||||
|
||||
@@ -1,46 +1,36 @@
|
||||
# The toolchain installer image. It does NOT run the cluster — it installs a toolchain
|
||||
# onto the host and gets out of the way.
|
||||
#
|
||||
# This exists to kill a bootstrap paradox: a plain bash installer needs curl, jq
|
||||
# and sha256sum to already be present, and a minimal Debian has none of them.
|
||||
# It carries its own toolchain, so the only host prerequisite is Docker.
|
||||
#
|
||||
# Two variants from one file:
|
||||
# Toolchain installer image: installs the pinned toolchain onto the host; Docker is the only prerequisite.
|
||||
# docker build -f ctrl/Dockerfile.deps --target deps -t <slug>-deps .
|
||||
# docker build -f ctrl/Dockerfile.deps --target deps-full -t <slug>-deps:full .
|
||||
#
|
||||
# deps-full bakes every pinned binary in at build time. `docker save` it and
|
||||
# you have the whole installer as one file to carry into an air-gapped network.
|
||||
# Notes: docs/notes/Dockerfile.deps.md
|
||||
|
||||
FROM debian:trixie-slim AS deps
|
||||
|
||||
# ca-certificates + curl: fetch and verify. graphviz + python3: render diagrams
|
||||
# and validate the arch model, so the host never needs an apt package.
|
||||
#
|
||||
# docker-cli, NOT docker.io: we only ever talk to the host's daemon through the
|
||||
# mounted socket, and under --no-install-recommends the docker.io package ships
|
||||
# docker-init without the actual `docker` binary.
|
||||
# curl: fetch and verify; graphviz + python3: diagrams. docker-cli, NOT docker.io
|
||||
# (which lacks the `docker` binary under --no-install-recommends).
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
ca-certificates curl jq graphviz python3 docker-cli \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# The installer is the generated one-file standalone kit, pins frozen in.
|
||||
ARG PROFILE=default
|
||||
WORKDIR /work
|
||||
COPY ctrl/versions.env /work/ctrl/versions.env
|
||||
COPY ctrl/deps.sh /work/ctrl/deps.sh
|
||||
RUN chmod +x /work/ctrl/deps.sh
|
||||
COPY standalone/${PROFILE}/rigdeps.sh /work/rigdeps.sh
|
||||
RUN chmod +x /work/rigdeps.sh
|
||||
|
||||
# Defaults; every one is overridable with -e at run time.
|
||||
ENV DEPS_SOURCE=upstream \
|
||||
OUT_BIN=/out/bin \
|
||||
HOST_ROOT=/host
|
||||
|
||||
ENTRYPOINT ["/work/ctrl/deps.sh"]
|
||||
ENTRYPOINT ["/work/rigdeps.sh"]
|
||||
CMD ["install"]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# deps-full — same image, binaries baked in, works with no network at all.
|
||||
# deps-full — same image, binaries and the addons' manifests baked in, works with no network.
|
||||
FROM deps AS deps-full
|
||||
RUN /work/ctrl/deps.sh fetch --to /opt/rig/bin
|
||||
RUN /work/rigdeps.sh fetch --to /opt/rig/bin \
|
||||
&& /work/rigdeps.sh manifests --to /opt/rig/manifests
|
||||
ENV DEPS_SOURCE=baked \
|
||||
BAKED_BIN=/opt/rig/bin
|
||||
BAKED_BIN=/opt/rig/bin \
|
||||
BAKED_MANIFESTS=/opt/rig/manifests
|
||||
|
||||
67
rig/ctrl/Tiltfile
Normal file
67
rig/ctrl/Tiltfile
Normal file
@@ -0,0 +1,67 @@
|
||||
# The dev loop, rig's half. `make tilt` from rig's folder, or from an overlay's forwarder.
|
||||
# rig owns this file: who we are, the context guard, the registry, the overlay's manifests.
|
||||
# The workload's half is the overlay's own Tiltfile, included at the end; edit that one.
|
||||
# Notes: docs/notes/Tiltfile.md
|
||||
|
||||
# ── who we are, and on which ports ─────────────────────────────────────────
|
||||
# Asked of ctrl/ports.sh (via lib/config.sh), never recomputed here in Starlark.
|
||||
_facts = str(local('bash ports.sh active', quiet=True)).split()
|
||||
CLUSTER = _facts[0]
|
||||
CTX = _facts[1]
|
||||
HTTP = _facts[2]
|
||||
HTTPS = _facts[3]
|
||||
TILT = _facts[4]
|
||||
REGISTRY = _facts[5]
|
||||
# Absolute paths, or '-' when there is none.
|
||||
MANIFESTS = '' if _facts[6] == '-' else _facts[6]
|
||||
OVERLAY = '' if _facts[7] == '-' else _facts[7]
|
||||
|
||||
# ── refuse to deploy into the wrong cluster ────────────────────────────────
|
||||
# Tilt fixes the context before parsing this file, so it can only be refused here.
|
||||
# `make tilt` passes --context; this catches a bare `tilt up`.
|
||||
allow_k8s_contexts(CTX)
|
||||
if k8s_context() != CTX:
|
||||
fail("Wrong kubectl context: '%s'. This is %s — run: make tilt, or tilt up --context %s"
|
||||
% (k8s_context(), CLUSTER, CTX))
|
||||
|
||||
# ── images go to this environment's own registry ───────────────────────────
|
||||
# Fail closed: name the registry rather than let Tilt infer it, or a miss pushes
|
||||
# an unqualified image to docker.io.
|
||||
default_registry('localhost:' + REGISTRY)
|
||||
|
||||
# ── the overlay's manifests ────────────────────────────────────────────────
|
||||
# Every namespace they use must exist before anything lands in it, and kustomize
|
||||
# does not order resources, so create them here (idempotent). The Namespaces they
|
||||
# declare are grouped as 'infra', whatever they are named.
|
||||
if MANIFESTS:
|
||||
_yaml = kustomize(MANIFESTS)
|
||||
k8s_yaml(_yaml)
|
||||
_declared = []
|
||||
_namespaces = {}
|
||||
for _o in decode_yaml_stream(_yaml):
|
||||
if not _o:
|
||||
continue
|
||||
_md = _o.get('metadata') or {}
|
||||
if _o.get('kind') == 'Namespace':
|
||||
_declared.append(_md.get('name'))
|
||||
_namespaces[_md.get('name')] = True
|
||||
elif _md.get('namespace'):
|
||||
_namespaces[_md.get('namespace')] = True
|
||||
for _ns in sorted(_namespaces.keys()):
|
||||
local('kubectl --context %s create namespace %s --dry-run=client -o yaml | kubectl --context %s apply -f -'
|
||||
% (CTX, _ns, CTX), quiet=True)
|
||||
if _declared:
|
||||
k8s_resource(objects=[_n + ':namespace' for _n in _declared], new_name='infra')
|
||||
|
||||
# ── the workload's half: the overlay's Tiltfile ────────────────────────────
|
||||
# Included, so its relative paths resolve from the overlay's own folder. It reads
|
||||
# these facts with os.getenv and never needs a path back into rig.
|
||||
os.putenv('RIG_CLUSTER', CLUSTER)
|
||||
os.putenv('RIG_CONTEXT', CTX)
|
||||
os.putenv('RIG_HTTP_PORT', HTTP)
|
||||
os.putenv('RIG_HTTPS_PORT', HTTPS)
|
||||
os.putenv('RIG_TILT_PORT', TILT)
|
||||
os.putenv('RIG_REGISTRY', 'localhost:' + REGISTRY)
|
||||
os.putenv('RIG_OVERLAY_DIR', OVERLAY)
|
||||
if OVERLAY and os.path.exists(OVERLAY + '/Tiltfile'):
|
||||
include(OVERLAY + '/Tiltfile')
|
||||
@@ -1,35 +1,59 @@
|
||||
#!/usr/bin/env bash
|
||||
# Install the addons the active profile asked for, in the order listed.
|
||||
# Each addon is its own idempotent script in ctrl/addons/ — adding one is adding
|
||||
# a file, not editing a dispatcher.
|
||||
#
|
||||
# Install the addons the configuration asks for (ADDONS), in the order listed.
|
||||
# One idempotent script per addon: the overlay's addons/<name>.sh first, then rig's ctrl/addons/.
|
||||
# Usage: addons.sh install | list
|
||||
# Notes: docs/notes/addons.md
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")"
|
||||
|
||||
source ./lib/config.sh
|
||||
load_config
|
||||
|
||||
# Every addon runs from here, wherever its file lives, so it can source ./lib/config.sh.
|
||||
export RIG_CTRL="$PWD"
|
||||
|
||||
# The script for one addon name: the overlay's, else rig's; empty if neither.
|
||||
addon_path() {
|
||||
local ov=""
|
||||
if [ -n "$OVERLAY_DIR" ]; then ov="$(_from_ctrl "$OVERLAY_DIR")/addons/$1.sh"; fi
|
||||
if [ -n "$ov" ] && [ -f "$ov" ]; then
|
||||
echo "$ov"
|
||||
elif [ -f "addons/$1.sh" ]; then
|
||||
echo "addons/$1.sh"
|
||||
fi
|
||||
}
|
||||
|
||||
install() {
|
||||
if [ -z "${ADDONS// /}" ]; then
|
||||
echo "no addons in profile '$PROFILE_NAME'"
|
||||
echo "no addons asked for (ADDONS is empty)"
|
||||
return
|
||||
fi
|
||||
local a
|
||||
local a p
|
||||
for a in $ADDONS; do
|
||||
if [ ! -f "addons/${a}.sh" ]; then
|
||||
echo "no such addon: addons/${a}.sh" >&2
|
||||
p=$(addon_path "$a")
|
||||
if [ -z "$p" ]; then
|
||||
echo "no such addon: $a (looked in the overlay's addons/ and ctrl/addons/)" >&2
|
||||
exit 1
|
||||
fi
|
||||
echo "addon: $a"
|
||||
bash "addons/${a}.sh"
|
||||
bash "$p"
|
||||
done
|
||||
}
|
||||
|
||||
list() {
|
||||
echo "profile '$PROFILE_NAME' wants: ${ADDONS:-none}"
|
||||
echo "wanted: ${ADDONS:-none}"
|
||||
echo "available:"
|
||||
ls addons/*.sh 2>/dev/null | xargs -n1 basename | sed 's/\.sh$//' | sed 's/^/ /'
|
||||
local f
|
||||
if [ -n "$OVERLAY_DIR" ]; then
|
||||
for f in "$(_from_ctrl "$OVERLAY_DIR")"/addons/*.sh; do
|
||||
[ -e "$f" ] || continue
|
||||
printf ' %-16s overlay\n' "$(basename "$f" .sh)"
|
||||
done
|
||||
fi
|
||||
for f in addons/*.sh; do
|
||||
[ -e "$f" ] || continue
|
||||
printf ' %-16s rig\n' "$(basename "$f" .sh)"
|
||||
done
|
||||
}
|
||||
|
||||
case "${1:-list}" in
|
||||
|
||||
@@ -1,12 +1,8 @@
|
||||
#!/usr/bin/env bash
|
||||
# cert-manager plus a self-signed cluster issuer.
|
||||
#
|
||||
# In a regulated estate almost everything is TLS, so the interesting question
|
||||
# during onboarding is "does this service present a cert my client trusts" — not
|
||||
# "can I reach a public ACME server". A local CA answers that offline, which is
|
||||
# also what makes the air-gapped profile usable.
|
||||
# cert-manager plus a self-signed cluster issuer (offline local CA).
|
||||
# Notes: docs/notes/addons.md
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")/.."
|
||||
cd "${RIG_CTRL:-$(dirname "$0")/..}"
|
||||
|
||||
source ./lib/config.sh
|
||||
load_config
|
||||
@@ -16,7 +12,9 @@ K="kubectl --context ${KUBECONTEXT}"
|
||||
if $K get deployment -n cert-manager cert-manager >/dev/null 2>&1; then
|
||||
echo " already installed"
|
||||
else
|
||||
$K apply -f "https://github.com/cert-manager/cert-manager/releases/download/${CERT_MANAGER_VERSION}/cert-manager.yaml"
|
||||
# The pinned manifest, verified on disk — never a URL applied directly.
|
||||
manifest=$(bash ./deps.sh manifest CERT_MANAGER)
|
||||
$K apply -f "$manifest"
|
||||
fi
|
||||
|
||||
echo " waiting for cert-manager..."
|
||||
|
||||
@@ -1,17 +1,9 @@
|
||||
#!/usr/bin/env bash
|
||||
# MetalLB — makes `Service type: LoadBalancer` actually get an address.
|
||||
#
|
||||
# Why it matters here: real manifests use LoadBalancer, because a real cluster
|
||||
# has one. On a bare kind cluster those Services sit at EXTERNAL-IP <pending>
|
||||
# forever with no error anywhere — the deployment looks fine and simply is not
|
||||
# reachable. Without this, every such Service has to be edited to NodePort,
|
||||
# which means the local manifests stop matching the ones being modelled.
|
||||
#
|
||||
# The address pool is derived from the kind Docker network at install time, not
|
||||
# hardcoded: Docker picks that subnet, it differs between machines, and a pool
|
||||
# outside it is silently unroutable.
|
||||
# The pool is derived from the kind Docker network at install time.
|
||||
# Notes: docs/notes/addons.md
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")/.."
|
||||
cd "${RIG_CTRL:-$(dirname "$0")/..}"
|
||||
|
||||
source ./lib/config.sh
|
||||
load_config
|
||||
@@ -54,13 +46,12 @@ echo " kind network $subnet → pool ${pool_start}-${pool_end}"
|
||||
if $K get deployment -n metallb-system controller >/dev/null 2>&1; then
|
||||
echo " already installed"
|
||||
else
|
||||
$K apply -f "https://raw.githubusercontent.com/metallb/metallb/${METALLB_VERSION}/config/manifests/metallb-native.yaml"
|
||||
# The pinned manifest, verified on disk — never a URL applied directly.
|
||||
manifest=$(bash ./deps.sh manifest METALLB)
|
||||
$K apply -f "$manifest"
|
||||
fi
|
||||
|
||||
# `kubectl wait` on a selector errors out immediately when nothing matches yet,
|
||||
# and right after apply the ReplicaSet has not created the pod — so it loses a
|
||||
# race it looks like it should win. `rollout status` waits for the Deployment
|
||||
# itself and handles the not-yet-created case.
|
||||
# `rollout status`, not `kubectl wait`: wait errors out while the pod doesn't exist yet.
|
||||
echo " waiting for the controller..."
|
||||
$K rollout status deployment/controller -n metallb-system --timeout=240s
|
||||
$K rollout status daemonset/speaker -n metallb-system --timeout=240s
|
||||
|
||||
@@ -1,12 +1,8 @@
|
||||
#!/usr/bin/env bash
|
||||
# metrics-server — makes `kubectl top` work.
|
||||
#
|
||||
# kind nodes serve kubelet metrics over a self-signed cert, so the standard
|
||||
# manifest never becomes ready without --kubelet-insecure-tls. That is fine here
|
||||
# (it is a local cluster) and is the single most common reason metrics-server
|
||||
# sits at 0/1 on kind.
|
||||
# metrics-server — makes `kubectl top` work (patched with --kubelet-insecure-tls for kind).
|
||||
# Notes: docs/notes/addons.md
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")/.."
|
||||
cd "${RIG_CTRL:-$(dirname "$0")/..}"
|
||||
|
||||
source ./lib/config.sh
|
||||
load_config
|
||||
@@ -14,7 +10,9 @@ load_config
|
||||
K="kubectl --context ${KUBECONTEXT}"
|
||||
|
||||
if ! $K get deployment -n kube-system metrics-server >/dev/null 2>&1; then
|
||||
$K apply -f "https://github.com/kubernetes-sigs/metrics-server/releases/download/${METRICS_SERVER_VERSION}/components.yaml"
|
||||
# The pinned manifest, verified on disk — never a URL applied directly.
|
||||
manifest=$(bash ./deps.sh manifest METRICS_SERVER)
|
||||
$K apply -f "$manifest"
|
||||
fi
|
||||
|
||||
$K patch deployment metrics-server -n kube-system --type=json \
|
||||
|
||||
@@ -1,64 +1,36 @@
|
||||
#!/usr/bin/env bash
|
||||
# Readiness check: is this machine ready to run rig?
|
||||
#
|
||||
# Reports and instructs; never silently fixes anything. Everything it finds is
|
||||
# either already fine, or something a human has to decide on.
|
||||
#
|
||||
# Runs ctrl/deps.sh host detection in a container when Docker is the only thing
|
||||
# installed, or directly when the toolchain is already present. Then adds the
|
||||
# checks that need this repo's config: profile sanity, CA trust, port clashes.
|
||||
# Readiness check: is this machine ready to run rig? Reports and instructs; never fixes.
|
||||
# Usage: check.sh [all | mem [status|push|all|backup|restore]] (all = every detail)
|
||||
# Notes: docs/notes/check.md
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")"
|
||||
|
||||
DEPS_IMAGE="${DEPS_IMAGE:-$(basename "$(cd .. && pwd)")-deps}"
|
||||
# `check mem` goes deeper on memory than the summary below: how far allocation
|
||||
# really climbs, and the WSL .wslconfig backup/restore.
|
||||
if [ "${1:-}" = mem ]; then
|
||||
shift
|
||||
exec bash ./mem.sh "${@:-status}"
|
||||
fi
|
||||
|
||||
# Host detection. Prefer running it bare — it needs no dependencies beyond
|
||||
# coreutils — and fall back to the container only if this shell can't.
|
||||
bash ./deps.sh detect
|
||||
# Compact by default: facts only with `all`; problems (!) always print.
|
||||
VERBOSE=""
|
||||
if [ "${1:-}" = all ]; then VERBOSE=1; fi
|
||||
fact() { if [ -n "$VERBOSE" ]; then echo "$@"; fi; }
|
||||
|
||||
# ── repo-level checks ──────────────────────────────────────────────────────
|
||||
bash ./deps.sh detect ${VERBOSE:+all}
|
||||
|
||||
source ./lib/config.sh
|
||||
load_config
|
||||
|
||||
echo
|
||||
echo "config"
|
||||
echo " profile ${PROFILE_NAME} (nodes=${NODES} audit=${AUDIT})"
|
||||
echo " cluster ${CLUSTER} (context ${KUBECONTEXT})"
|
||||
echo " registry ${REGISTRY_MODE}"
|
||||
echo " ingress ${INGRESS_MODE}"
|
||||
|
||||
if [ ! -f ./.env ]; then
|
||||
echo " ! ctrl/.env missing — copy it: cp ctrl/.env.example ctrl/.env"
|
||||
fi
|
||||
|
||||
# ── memory ─────────────────────────────────────────────────────────────────
|
||||
#
|
||||
# A profile on a box that is already full is the most common first failure, and
|
||||
# it presents as pods stuck Pending rather than anything that says "memory".
|
||||
# Warns; never blocks. Whether to try anyway is the user's call.
|
||||
|
||||
# A /proc/meminfo field in MB, 0 if absent. MEMINFO and OVERCOMMIT_FILE exist
|
||||
# only so the tight and does-not-fit branches can be exercised against another
|
||||
# machine's real numbers; in normal use they are the kernel's own files.
|
||||
# A /proc/meminfo field in MB, 0 if absent. MEMINFO/OVERCOMMIT_FILE override for testing.
|
||||
mb_of() {
|
||||
awk -v k="$1:" '$1 == k { printf "%d", $2 / 1024; found = 1 }
|
||||
END { if (!found) printf "0" }' "${MEMINFO:-/proc/meminfo}"
|
||||
}
|
||||
|
||||
# What one node costs, measured rather than guessed. On 2026-09-11 a minimal
|
||||
# control-plane node ran at 620 MiB idle and ~728 MiB with a small mock, plus
|
||||
# 16 MiB for the local registry — ~745 MiB of working set. 800 rounds that up,
|
||||
# and agrees with the 800 MB observed independently on a larger rig. Worker
|
||||
# nodes carry no etcd or apiserver and are lighter, so for a multi-node shape
|
||||
# this errs high. It is the cluster alone: whatever you deploy comes on top.
|
||||
NODE_MB=800
|
||||
|
||||
# Every running container's working set in MB, tagged with the kind cluster it
|
||||
# belongs to ('-' when it is not kind). docker stats reports usage minus page
|
||||
# cache, which is what actually competes — cache is handed back under pressure.
|
||||
# Counting only kind would hide the usual culprit on a managed workspace, where
|
||||
# the memory is held by other containers entirely.
|
||||
# NODE_MB (cost of one node) comes from load_config in lib/config.sh; do not copy it here.
|
||||
|
||||
# Every running container's working set in MB, tagged with its kind cluster ('-' if none).
|
||||
container_mb() {
|
||||
docker info >/dev/null 2>&1 || return 0
|
||||
awk -F'\t' '
|
||||
@@ -84,6 +56,49 @@ container_mb() {
|
||||
<(docker stats --no-stream --format '{{.Name}}\t{{.MemUsage}}' 2>/dev/null)
|
||||
}
|
||||
|
||||
port_busy() {
|
||||
if command -v ss >/dev/null 2>&1; then
|
||||
ss -ltn "sport = :$1" 2>/dev/null | grep -q LISTEN && return 0 || return 1
|
||||
fi
|
||||
# iproute2 is absent from a minimal Debian, so fall back to procfs rather
|
||||
# than silently reporting everything as free.
|
||||
local hex; hex=$(printf ':%04X' "$1")
|
||||
grep -qi "^ *[0-9]*: [0-9A-F]*$hex " /proc/net/tcp /proc/net/tcp6 2>/dev/null
|
||||
}
|
||||
|
||||
echo
|
||||
echo "rig"
|
||||
echo " cluster ${CLUSTER} (${KUBECONTEXT}) profile ${PROFILE_NAME}, ${NODES} node(s), registry ${REGISTRY_MODE}"
|
||||
if [ -n "${OVERLAY:-}" ]; then
|
||||
echo " overlay $(basename "$(_abs_from_ctrl "$OVERLAY_DIR")") ($(_abs_from_ctrl "$OVERLAY_DIR"))"
|
||||
elif [ -n "$OVERLAY_DIR" ]; then
|
||||
fact " overlay none named — rig's own ${OVERLAY_DIR}"
|
||||
fi
|
||||
if [ -n "$VERBOSE" ] && [ -n "$OVERLAY_DIR" ]; then
|
||||
ov=$(_from_ctrl "$OVERLAY_DIR") pieces=""
|
||||
for piece in rig.env k8s/overlays/dev kind-config.yaml.tpl addons Tiltfile; do
|
||||
[ -e "$ov/$piece" ] && pieces+="$piece "
|
||||
done
|
||||
echo " provides: ${pieces:-nothing rig reads}"
|
||||
fi
|
||||
fact " manifests ${MANIFESTS_DIR:-none}"
|
||||
fact " kind config ${KIND_CONFIG}"
|
||||
fact " ingress ${INGRESS_MODE}"
|
||||
if [ ! -f ./.env ]; then
|
||||
fact " .env none — built-in defaults (cp ctrl/.env.example ctrl/.env to set values)"
|
||||
fi
|
||||
if [ -n "${STALE_MANIFESTS_DIR:-}" ]; then
|
||||
echo " ! .env MANIFESTS_DIR=${STALE_MANIFESTS_DIR} is the old default; rig's examples moved"
|
||||
echo " to examples/ — delete that line from ctrl/.env (ignored until then)"
|
||||
fi
|
||||
# registry.sh points containerd at certs.d, which only works if the kind config says so,
|
||||
# and a kind config is fixed at creation: a project's own file that drops it fails silently.
|
||||
if [ "$REGISTRY_MODE" != none ] && ! grep -q 'config_path *= *"/etc/containerd/certs.d"' "$KIND_CONFIG"; then
|
||||
echo " ! kind ${KIND_CONFIG} lacks the containerd config_path patch that registry mode"
|
||||
echo " '${REGISTRY_MODE}' needs — copy it from ctrl/k8s/kind-config.yaml.tpl"
|
||||
fi
|
||||
|
||||
# ── memory: does this cluster fit right now? Warns; never blocks. ──────────
|
||||
total_mb=$(mb_of MemTotal)
|
||||
avail_mb=$(mb_of MemAvailable)
|
||||
swap_used_mb=$(( $(mb_of SwapTotal) - $(mb_of SwapFree) ))
|
||||
@@ -91,131 +106,111 @@ overcommit=$(cat "${OVERCOMMIT_FILE:-/proc/sys/vm/overcommit_memory}" 2>/dev/nul
|
||||
need_mb=$(( NODES * NODE_MB ))
|
||||
|
||||
rows=$(container_mb)
|
||||
# Once this environment's own cluster is running, its real footprint is already
|
||||
# out of MemAvailable and the per-node estimate stops being relevant. Subtracting
|
||||
# the measurement from the estimate would count the same memory twice, and a
|
||||
# running cluster that happens to sit under 800 MB would still "need" the gap.
|
||||
# If our cluster is already up, its memory is already out of MemAvailable: need nothing more.
|
||||
ours_mb=$(awk -F'\t' -v c="$CLUSTER" '$2 == c { s += $1 } END { print s + 0 }' <<< "$rows")
|
||||
still_mb=$(( ours_mb > 0 ? 0 : need_mb ))
|
||||
headroom=$(( avail_mb - still_mb ))
|
||||
|
||||
echo
|
||||
echo "memory"
|
||||
printf " this profile ~%d MB %s node(s) x %d MB — the cluster alone, your workload on top\n" \
|
||||
"$need_mb" "$NODES" "$NODE_MB"
|
||||
if [ "$ours_mb" -gt 0 ]; then
|
||||
printf " already held %d MB by '%s', which is up\n" "$ours_mb" "$CLUSTER"
|
||||
fi
|
||||
printf " available %d MB of %d MB\n" "$avail_mb" "$total_mb"
|
||||
|
||||
# The biggest things holding memory right now, other than this cluster: kind
|
||||
# clusters summed per cluster, everything else by container name.
|
||||
# The biggest things holding memory, other than this cluster: kind clusters summed, the rest by name.
|
||||
others=$(awk -F'\t' -v c="$CLUSTER" '
|
||||
$2 != c && $2 != "-" && $2 != "" { k["kind cluster \x27" $2 "\x27"] += $1 }
|
||||
$2 == "-" { k["container \x27" $3 "\x27"] += $1 }
|
||||
END { for (n in k) printf "%d\t%s\n", k[n], n }' <<< "$rows" | sort -rn)
|
||||
if [ -n "$others" ]; then
|
||||
echo " held elsewhere:"
|
||||
head -6 <<< "$others" | awk -F'\t' '{ printf " %6d MB %s\n", $1, $2 }'
|
||||
n_others=$(wc -l <<< "$others")
|
||||
if [ "$n_others" -gt 6 ]; then
|
||||
echo " ... and $((n_others - 6)) more"
|
||||
fi
|
||||
fi
|
||||
|
||||
headroom=$(( avail_mb - still_mb ))
|
||||
if [ "$still_mb" -eq 0 ]; then
|
||||
if [ "$headroom" -ge 512 ]; then
|
||||
printf " fits — already up; %d MB headroom for what you deploy\n" "$headroom"
|
||||
else
|
||||
printf " ! already up, but only %d MB headroom for anything you deploy\n" "$headroom"
|
||||
fi
|
||||
if [ "$still_mb" -eq 0 ] && [ "$headroom" -ge 512 ]; then
|
||||
printf " memory up, holding %d MB — %d MB headroom for what you deploy\n" "$ours_mb" "$headroom"
|
||||
elif [ "$still_mb" -eq 0 ]; then
|
||||
printf " ! memory up, but only %d MB headroom for anything you deploy\n" "$headroom"
|
||||
elif [ "$headroom" -ge 512 ]; then
|
||||
printf " fits — %d MB headroom for what you deploy\n" "$headroom"
|
||||
printf " memory fits — ~%d MB for %s node(s), %d MB headroom\n" "$need_mb" "$NODES" "$headroom"
|
||||
elif [ "$headroom" -ge 0 ]; then
|
||||
printf " ! fits, but only %d MB headroom for anything you deploy\n" "$headroom"
|
||||
printf " ! memory fits, but only %d MB headroom (~%d MB for %s node(s))\n" "$headroom" "$need_mb" "$NODES"
|
||||
else
|
||||
printf " ! does not fit right now: ~%d MB needed, %d MB available\n" "$still_mb" "$avail_mb"
|
||||
printf " ! memory does not fit: ~%d MB needed, %d MB available\n" "$still_mb" "$avail_mb"
|
||||
# Two failures with opposite fixes, and telling them apart is the point.
|
||||
if [ "$still_mb" -le "$total_mb" ]; then
|
||||
echo " The machine is big enough; something else is holding memory (above)."
|
||||
echo " Stopping that is what helps — a bigger VM would not."
|
||||
echo " something else holds it (below) — stopping that helps, a bigger VM would not."
|
||||
if grep -q 'kind cluster' <<< "$others"; then
|
||||
echo " 'make cluster free' stops the other kind clusters. It stops, never deletes."
|
||||
fi
|
||||
else
|
||||
echo " The machine itself is too small: ~${still_mb} MB needed, ${total_mb} MB total."
|
||||
fi
|
||||
fi
|
||||
|
||||
if [ "$swap_used_mb" -gt 0 ]; then
|
||||
printf " ! %d MB already in swap — available memory does not count it, so expect a\n" "$swap_used_mb"
|
||||
echo " cluster here to be slow well before it fails"
|
||||
echo " the machine itself is too small: ${total_mb} MB total."
|
||||
fi
|
||||
if [ "$overcommit" = "1" ]; then
|
||||
echo " ! overcommit=1: allocations never fail here, so read 'fits' as a ceiling."
|
||||
echo " A cluster that starts cleanly can still lose processes to the OOM killer."
|
||||
fi
|
||||
|
||||
# The CA reaches three places and only one of them is ours. Report the other two.
|
||||
if [ -n "${REGISTRY_CA_FILE:-}" ]; then
|
||||
echo
|
||||
echo "registry CA"
|
||||
if [ ! -r "$REGISTRY_CA_FILE" ]; then
|
||||
echo " ! REGISTRY_CA_FILE not readable: $REGISTRY_CA_FILE"
|
||||
else
|
||||
echo " file $REGISTRY_CA_FILE"
|
||||
host="${REGISTRY_REMOTE_URL#*://}"; host="${host%%/*}"
|
||||
if [ -n "$host" ] && [ ! -f "/etc/docker/certs.d/${host}/ca.crt" ]; then
|
||||
echo " ! the HOST docker daemon does not trust it yet:"
|
||||
echo " sudo mkdir -p /etc/docker/certs.d/${host}"
|
||||
echo " sudo cp ${REGISTRY_CA_FILE} /etc/docker/certs.d/${host}/ca.crt"
|
||||
echo " (kind nodes are handled by registry.sh; in-cluster clients are the workload's job)"
|
||||
if [ -n "$others" ] && { [ -n "$VERBOSE" ] || [ "$headroom" -lt 512 ]; }; then
|
||||
echo " held elsewhere:"
|
||||
head -6 <<< "$others" | awk -F'\t' '{ printf " %6d MB %s\n", $1, $2 }'
|
||||
n_others=$(wc -l <<< "$others")
|
||||
if [ "$n_others" -gt 6 ]; then
|
||||
echo " ... and $((n_others - 6)) more"
|
||||
fi
|
||||
fi
|
||||
if [ "$swap_used_mb" -gt 0 ]; then
|
||||
fact " ${swap_used_mb} MB already in swap, which 'available' does not count: expect slow before failing"
|
||||
fi
|
||||
|
||||
# Host ports this environment will try to bind. Checked before cluster creation
|
||||
# because docker reports a clash halfway through, as an opaque
|
||||
# "failed to bind host port ...: address already in use".
|
||||
echo
|
||||
echo "ports (block derived from the directory name — see 'make ports')"
|
||||
|
||||
port_busy() {
|
||||
if command -v ss >/dev/null 2>&1; then
|
||||
ss -ltn "sport = :$1" 2>/dev/null | grep -q LISTEN && return 0 || return 1
|
||||
if [ "$overcommit" = "1" ]; then
|
||||
fact " overcommit=1: allocations never fail, so read 'fits' as a ceiling (OOM killer settles up)"
|
||||
fi
|
||||
# iproute2 is absent from a minimal Debian, so fall back to procfs rather
|
||||
# than silently reporting everything as free.
|
||||
local hex; hex=$(printf ':%04X' "$1")
|
||||
grep -qi "^ *[0-9]*: [0-9A-F]*$hex " /proc/net/tcp /proc/net/tcp6 2>/dev/null
|
||||
}
|
||||
|
||||
# A port held by THIS environment's own cluster is not a clash — it is the thing
|
||||
# working. Reporting it as a problem every time the cluster is up would train
|
||||
# people to ignore this section, which is the opposite of the point.
|
||||
# Extract with a second grep rather than `tr -d ':->'`: in tr, ':->' is the
|
||||
# character RANGE ':' to '>', which does not contain '-', so the trailing dash
|
||||
# survives and nothing ever matches.
|
||||
# ── ports: checked before creation; docker reports a clash only halfway through. ──
|
||||
# Ports held by our own cluster are not clashes. Second grep, not `tr -d ':->'` (a tr range).
|
||||
ours=$(docker ps --filter "label=io.x-k8s.kind.cluster=${CLUSTER}" \
|
||||
--format '{{.Ports}}' 2>/dev/null | tr ',' '\n' \
|
||||
| grep -oE ':[0-9]+->' | grep -oE '[0-9]+' || true)
|
||||
|
||||
clash=0
|
||||
clash=0 list="" mine=0
|
||||
for entry in "HTTP:${HTTP_PORT}" "HTTPS:${HTTPS_PORT}" \
|
||||
"TILT:${TILT_PORT}" "REGISTRY:${REGISTRY_PORT}"; do
|
||||
name="${entry%%:*}"; p="${entry#*:}"
|
||||
[ -n "$p" ] || continue
|
||||
list+="$p "
|
||||
if ! port_busy "$p"; then
|
||||
printf " %-9s %-6s free\n" "$name" "$p"
|
||||
fact "$(printf " %-9s %-6s free" "$name" "$p")"
|
||||
elif echo "$ours" | grep -qx "$p"; then
|
||||
printf " %-9s %-6s in use by this environment's cluster\n" "$name" "$p"
|
||||
mine=1
|
||||
fact "$(printf " %-9s %-6s in use by this environment's cluster" "$name" "$p")"
|
||||
else
|
||||
printf " ! %-9s %-6s IN USE by something else\n" "$name" "$p"
|
||||
printf " ! ports %s %s IN USE by something else\n" "$name" "$p"
|
||||
clash=1
|
||||
fi
|
||||
done
|
||||
|
||||
if [ "$clash" -eq 1 ]; then
|
||||
echo " override the clashing one in ctrl/.env, e.g. HTTP_PORT=21080"
|
||||
echo " (or rename this directory — the whole block follows the name)"
|
||||
echo " override it in ctrl/.env (e.g. HTTP_PORT=21080), or rename this directory"
|
||||
elif [ "$mine" -eq 1 ]; then
|
||||
echo " ports ${list% } held by this cluster"
|
||||
else
|
||||
echo " ports ${list% } free"
|
||||
fi
|
||||
if [ -n "${OVERLAY:-}" ]; then
|
||||
fact " derived from the overlay's folder name"
|
||||
else
|
||||
fact " derived from the directory name; pin them: bash ctrl/ports.sh persist"
|
||||
fi
|
||||
|
||||
# ── what `make cluster up` wires in beside the cluster ─────────────────────
|
||||
REG_NAME="${CLUSTER}-registry"
|
||||
if state=$(docker inspect -f '{{.State.Status}}' "$REG_NAME" 2>/dev/null); then
|
||||
echo " registry localhost:${REGISTRY_PORT} ($state)"
|
||||
else
|
||||
fact " registry no container yet — 'make cluster up' starts it"
|
||||
fi
|
||||
echo " addons ${ADDONS:-none}"
|
||||
if [ -n "$VERBOSE" ]; then
|
||||
bash ./addons.sh list | sed -n '3,$p' | sed 's/^/ /'
|
||||
fi
|
||||
|
||||
# The CA reaches three places and only one of them is ours. Report the other two.
|
||||
if [ -n "${REGISTRY_CA_FILE:-}" ]; then
|
||||
if [ ! -r "$REGISTRY_CA_FILE" ]; then
|
||||
echo " ! CA REGISTRY_CA_FILE not readable: $REGISTRY_CA_FILE"
|
||||
else
|
||||
fact " CA $REGISTRY_CA_FILE"
|
||||
host="${REGISTRY_REMOTE_URL#*://}"; host="${host%%/*}"
|
||||
if [ -n "$host" ] && [ ! -f "/etc/docker/certs.d/${host}/ca.crt" ]; then
|
||||
echo " ! CA the HOST docker daemon does not trust it yet:"
|
||||
echo " sudo mkdir -p /etc/docker/certs.d/${host}"
|
||||
echo " sudo cp ${REGISTRY_CA_FILE} /etc/docker/certs.d/${host}/ca.crt"
|
||||
echo " (kind nodes are handled by registry.sh; in-cluster clients are the workload's job)"
|
||||
fi
|
||||
fi
|
||||
fi
|
||||
|
||||
@@ -1,17 +1,7 @@
|
||||
#!/usr/bin/env bash
|
||||
# Cluster lifecycle, plus what else is running on this machine.
|
||||
#
|
||||
# `list` and `free` live here rather than in a separate script because a
|
||||
# near-identical second name (cluster / clusters) is a trap — you reach for one
|
||||
# and get the other. One target, one file, unambiguous subcommands.
|
||||
#
|
||||
# "Idempotent" here means CONVERGENT, not "exits early if the cluster exists".
|
||||
# That distinction matters: an interrupted first run can leave a cluster created
|
||||
# but not finished, and returning early on the re-run would strand it there.
|
||||
# The create step is conditional; every step after it always runs, and each one
|
||||
# is individually idempotent.
|
||||
#
|
||||
# Cluster lifecycle (convergent, not exit-early), plus what else runs on this machine.
|
||||
# Usage: cluster.sh up | down | reset | list | free
|
||||
# Notes: docs/notes/cluster.md
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")"
|
||||
|
||||
@@ -23,15 +13,15 @@ up() {
|
||||
echo "cluster '$CLUSTER' exists — converging"
|
||||
else
|
||||
# Say what this profile locks in BEFORE spending minutes building it:
|
||||
# the audit policy is an apiserver flag and cannot be changed later.
|
||||
# the kind config is fixed at creation and cannot be changed later.
|
||||
echo "creating cluster '$CLUSTER' from profile '$PROFILE_NAME'"
|
||||
echo " shape ${KIND_CONFIG_SHOWN}"
|
||||
echo " overlay ${OVERLAY_DIR:-none}"
|
||||
echo " kind config ${KIND_CONFIG}"
|
||||
echo " nodes $NODES"
|
||||
echo " image $NODE_IMAGE"
|
||||
echo " audit $AUDIT"
|
||||
echo " ingress $INGRESS_MODE"
|
||||
echo " registry $REGISTRY_MODE"
|
||||
echo " (audit is fixed at creation — 'make cluster reset' to change it)"
|
||||
echo " (fixed at creation — edit the kind config, then 'make cluster reset')"
|
||||
echo
|
||||
|
||||
render_kind_config | kind create cluster --config -
|
||||
@@ -69,7 +59,7 @@ down() {
|
||||
}
|
||||
|
||||
# The escape hatch for a wedged cluster, and the only way to change a
|
||||
# creation-time setting such as the audit policy.
|
||||
# creation-time setting such as the node count or port mappings.
|
||||
reset() {
|
||||
down
|
||||
echo
|
||||
|
||||
513
rig/ctrl/deps.sh
513
rig/ctrl/deps.sh
@@ -1,33 +1,21 @@
|
||||
#!/usr/bin/env bash
|
||||
# Toolchain installer: detect the host, install a pinned toolchain onto it, then
|
||||
# report what it could not do.
|
||||
#
|
||||
# It never runs the cluster, never uses sudo or apt, and writes only into
|
||||
# $OUT_BIN (default ~/.local/bin). Everything that would touch the host proper —
|
||||
# systemd, inotify limits, .wslconfig, docker group — is REPORTED for a human to
|
||||
# decide on, never performed. That is what makes it safe to run on a machine that
|
||||
# already has a working setup.
|
||||
#
|
||||
# Usage (normally via `make deps`, or directly):
|
||||
# deps.sh detect # report host facts only, change nothing
|
||||
# deps.sh fetch [core|dev] [--to DIR] # download + verify into DIR
|
||||
# deps.sh install [core|dev] # detect, fetch, install, report
|
||||
#
|
||||
# Tiers: 'core' is kubectl + jq (talk to a cluster); 'dev' adds kind and tilt
|
||||
# Default is dev.
|
||||
#
|
||||
# Runs both inside the installer container and bare on a host. Inside the
|
||||
# container, host files are read through $HOST_ROOT (mount / as :ro); bare, it
|
||||
# falls back to /.
|
||||
# rig:standalone rigdeps detect
|
||||
# Toolchain installer: detect the host, install pinned tools into $OUT_BIN, report
|
||||
# host actions it will not perform (no sudo, no apt). Usually via `make deps`.
|
||||
# Usage: deps.sh [detect [all] | list | verify [core|dev] | fetch [core|dev] [--to DIR] | install [core|dev]
|
||||
# | manifest NAME | manifests [--to DIR]]
|
||||
# Notes: docs/notes/deps.md
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
# Keep the caller's cwd so a relative --to resolves where the user expects,
|
||||
# not against ctrl/ once we've moved.
|
||||
# Keep the caller's cwd so a relative --to resolves there, not against ctrl/.
|
||||
INVOKED_FROM="$PWD"
|
||||
cd "$(dirname "$0")"
|
||||
|
||||
source ./versions.env
|
||||
# Pins arrive through load_config, not by sourcing versions.env, so `make
|
||||
# standalone` can freeze them in.
|
||||
source ./lib/config.sh
|
||||
load_config
|
||||
|
||||
# Resolve a possibly-relative path against the caller's original directory.
|
||||
abspath() {
|
||||
@@ -46,12 +34,11 @@ BAKED_BIN="${BAKED_BIN:-/opt/rig/bin}"
|
||||
# Collected by detect(), printed by report_manual() at the very end.
|
||||
MANUAL=()
|
||||
|
||||
# Host FILES (/etc/..., /mnt/c/...) must be read through the mount. Kernel-level
|
||||
# facts (kernel version, meminfo, inotify) are shared with the container, so the
|
||||
# container's own view is already the host's.
|
||||
# A /proc/meminfo field in MB, 0 if the field is absent. MEMINFO exists so the
|
||||
# tight and does-not-fit branches can be exercised against a real machine's
|
||||
# numbers from somewhere else; in normal use it is always /proc/meminfo.
|
||||
# Facts print only with VERBOSE (`detect all`); problems (! and -) always print.
|
||||
fact() { if [ -n "${VERBOSE:-}" ]; then echo "$@"; fi; }
|
||||
|
||||
# Host FILES are read through $HOST_ROOT; kernel facts are shared with the container.
|
||||
# A /proc/meminfo field in MB, 0 if absent. MEMINFO overrides the source for testing.
|
||||
mb_of() {
|
||||
awk -v k="$1:" '$1 == k { printf "%d", $2 / 1024; found = 1 }
|
||||
END { if (!found) printf "0" }' "${MEMINFO:-/proc/meminfo}"
|
||||
@@ -66,11 +53,90 @@ host_file() {
|
||||
fi
|
||||
}
|
||||
|
||||
# ── the tools this script itself needs ─────────────────────────────────────
|
||||
|
||||
arch() {
|
||||
case "$(uname -m)" in
|
||||
x86_64|amd64) echo amd64 ;;
|
||||
aarch64|arm64) echo arm64 ;;
|
||||
*) uname -m ;;
|
||||
esac
|
||||
}
|
||||
|
||||
# Pins are amd64 only: refuse elsewhere and print how to get the right checksums.
|
||||
require_amd64() {
|
||||
local a; a=$(arch)
|
||||
[ "$a" = "amd64" ] && return 0
|
||||
cat >&2 <<EOF
|
||||
This machine is ${a} ($(uname -m)); every pin in this script is linux/amd64.
|
||||
|
||||
Nothing here would run, so it does not download. To make an ${a} version, the
|
||||
URLs need the ${a} artifact and the checksums need to come from each project's
|
||||
own published list — not from these values, and not from a download you did:
|
||||
|
||||
curl -sSL https://github.com/kubernetes-sigs/kind/releases/download/${KIND_VERSION}/checksums.txt
|
||||
curl -sSL https://dl.k8s.io/release/${KUBECTL_VERSION}/bin/linux/${a}/kubectl.sha256
|
||||
curl -sSL https://github.com/tilt-dev/tilt/releases/download/v${TILT_VERSION}/checksums.txt
|
||||
curl -sSL https://github.com/tilt-dev/ctlptl/releases/download/v${CTLPTL_VERSION}/checksums.txt
|
||||
curl -sSL https://github.com/jqlang/jq/releases/download/jq-${JQ_VERSION}/sha256sum.txt
|
||||
|
||||
Edit the pinned block at the top of this file with what those print.
|
||||
EOF
|
||||
exit 1
|
||||
}
|
||||
|
||||
DL=""
|
||||
pick_downloader() {
|
||||
if command -v curl >/dev/null 2>&1; then DL=curl
|
||||
elif command -v wget >/dev/null 2>&1; then DL=wget
|
||||
else
|
||||
echo "neither curl nor wget is installed, so nothing can be downloaded." >&2
|
||||
echo "Install one first: $(pkg_install_cmd curl)" >&2
|
||||
exit 1
|
||||
fi
|
||||
}
|
||||
|
||||
download() {
|
||||
local url="$1" out="$2"
|
||||
case "$DL" in
|
||||
curl) curl -fsSL --retry 3 -o "$out" "$url" ;;
|
||||
wget) wget -q --tries=3 -O "$out" "$url" ;;
|
||||
esac
|
||||
}
|
||||
|
||||
SHA=""
|
||||
pick_sha() {
|
||||
if command -v sha256sum >/dev/null 2>&1; then SHA=sha256sum
|
||||
elif command -v shasum >/dev/null 2>&1; then SHA="shasum -a 256"
|
||||
else
|
||||
echo "no sha256sum and no shasum — downloads could not be verified." >&2
|
||||
echo "Refusing to install unverified binaries." >&2
|
||||
exit 1
|
||||
fi
|
||||
}
|
||||
|
||||
# ── package manager, for the instructions only ─────────────────────────────
|
||||
# Never runs one; names the right one so reported actions are pasteable.
|
||||
|
||||
pkg_install_cmd() {
|
||||
local pkg="$1"
|
||||
if command -v apt-get >/dev/null 2>&1; then echo "sudo apt-get update && sudo apt-get install -y $pkg"
|
||||
elif command -v dnf >/dev/null 2>&1; then echo "sudo dnf install -y $pkg"
|
||||
elif command -v yum >/dev/null 2>&1; then echo "sudo yum install -y $pkg"
|
||||
elif command -v zypper >/dev/null 2>&1; then echo "sudo zypper install -y $pkg"
|
||||
elif command -v apk >/dev/null 2>&1; then echo "sudo apk add $pkg"
|
||||
else echo "install '$pkg' with this system's package manager"
|
||||
fi
|
||||
}
|
||||
|
||||
docker_pkg() {
|
||||
# Debian and Ubuntu call it docker.io; the RPM distros call it docker.
|
||||
if command -v apt-get >/dev/null 2>&1; then echo docker.io; else echo docker; fi
|
||||
}
|
||||
|
||||
# ── detect ─────────────────────────────────────────────────────────────────
|
||||
|
||||
# Windows outside WSL — Git Bash, MSYS, Cygwin — looks close enough to work and
|
||||
# then fails in a pile of confusing ways: no /proc, no docker socket, none of
|
||||
# the tooling. Detectable, so name it instead.
|
||||
# Windows outside WSL (Git Bash, MSYS, Cygwin) fails confusingly; name it instead.
|
||||
require_linux() {
|
||||
case "$(uname -s)" in
|
||||
MINGW*|MSYS*|CYGWIN*)
|
||||
@@ -95,34 +161,32 @@ is_wsl() { grep -qi microsoft /proc/version 2>/dev/null; }
|
||||
|
||||
detect() {
|
||||
echo "host"
|
||||
echo " kernel $(uname -r)"
|
||||
fact " kernel $(uname -r)"
|
||||
|
||||
local osr; osr=$(host_file /etc/os-release)
|
||||
[ -r "$osr" ] && echo " distro $(sed -n 's/^PRETTY_NAME="\(.*\)"/\1/p' "$osr")"
|
||||
local osr distro=""; osr=$(host_file /etc/os-release)
|
||||
[ -r "$osr" ] && distro=$(sed -n 's/^PRETTY_NAME="\(.*\)"/\1/p' "$osr")
|
||||
echo " distro ${distro:-unknown} $(arch), $(if is_wsl; then echo WSL; else echo native linux; fi)"
|
||||
|
||||
# In MB. Whole gigabytes lose nearly half a GB on exactly the machines where
|
||||
# it matters: 1874 MB available used to print as "1 GB". Facts only — whether
|
||||
# that is enough depends on the profile, which check.sh knows and this does not.
|
||||
# In MB (whole GB rounds away too much). Facts only; check.sh judges sufficiency.
|
||||
local total_mb avail_mb swap_total_mb swap_used_mb om
|
||||
total_mb=$(mb_of MemTotal)
|
||||
avail_mb=$(mb_of MemAvailable)
|
||||
swap_total_mb=$(mb_of SwapTotal)
|
||||
swap_used_mb=$(( swap_total_mb - $(mb_of SwapFree) ))
|
||||
printf " memory %d MB total, %d MB available\n" "$total_mb" "$avail_mb"
|
||||
if [ "$swap_total_mb" -gt 0 ]; then
|
||||
printf " swap %d MB used of %d MB\n" "$swap_used_mb" "$swap_total_mb"
|
||||
fi
|
||||
printf " memory %d MB total, %d MB available%s\n" "$total_mb" "$avail_mb" \
|
||||
"$(if [ "$swap_used_mb" -gt 0 ]; then echo ", $swap_used_mb MB in swap"; fi)"
|
||||
|
||||
# How the kernel answers an allocation it cannot really satisfy. With 1 it
|
||||
# always says yes and settles up later with the OOM killer, so a cluster that
|
||||
# starts cleanly can still lose processes afterwards.
|
||||
# Overcommit mode: with 1 the OOM killer settles up later, after a clean start.
|
||||
om=$(cat "${OVERCOMMIT_FILE:-/proc/sys/vm/overcommit_memory}" 2>/dev/null || echo '?')
|
||||
case "$om" in
|
||||
0) echo " overcommit 0 heuristic — allocations are granted on a guess" ;;
|
||||
1) echo " overcommit 1 always — every allocation succeeds; the OOM killer is the only limit" ;;
|
||||
2) echo " overcommit 2 strict — an allocation fails honestly instead of killing later" ;;
|
||||
0) fact " overcommit 0 heuristic — allocations are granted on a guess" ;;
|
||||
1) fact " overcommit 1 always — every allocation succeeds; the OOM killer is the only limit" ;;
|
||||
2) fact " overcommit 2 strict — an allocation fails honestly instead of killing later" ;;
|
||||
esac
|
||||
|
||||
fact " install to $OUT_BIN"
|
||||
detect_libc
|
||||
detect_prereqs
|
||||
detect_wsl
|
||||
detect_filesystem
|
||||
detect_docker
|
||||
@@ -132,18 +196,13 @@ detect() {
|
||||
|
||||
detect_wsl() {
|
||||
if ! is_wsl; then
|
||||
echo " platform native linux"
|
||||
return
|
||||
fi
|
||||
|
||||
echo " platform WSL"
|
||||
|
||||
# systemd is off by default in WSL, and the ingress/DNS paths that use a
|
||||
# host service need it. Enabling it requires a Windows-side restart, which
|
||||
# cannot be issued from inside the distro.
|
||||
# systemd is off by default in WSL; enabling it needs a Windows-side restart.
|
||||
local wc; wc=$(host_file /etc/wsl.conf)
|
||||
if [ -r "$wc" ] && grep -qE '^\s*systemd\s*=\s*true' "$wc"; then
|
||||
echo " systemd enabled in wsl.conf"
|
||||
fact " systemd enabled in wsl.conf"
|
||||
else
|
||||
echo " ! systemd not enabled in /etc/wsl.conf"
|
||||
MANUAL+=("Enable systemd — add to /etc/wsl.conf:
|
||||
@@ -155,27 +214,24 @@ detect_wsl() {
|
||||
# WSL regenerates /etc/resolv.conf on every boot, which silently reverts any
|
||||
# local DNS setup.
|
||||
if [ -r "$wc" ] && grep -qE '^\s*generateResolvConf\s*=\s*false' "$wc"; then
|
||||
echo " resolv.conf pinned (generateResolvConf=false)"
|
||||
fact " resolv.conf pinned (generateResolvConf=false)"
|
||||
else
|
||||
echo " - resolv.conf is WSL-generated; DNS_MODE=dnsmasq would be reverted on reboot"
|
||||
fact " - resolv.conf is WSL-generated; DNS_MODE=dnsmasq would be reverted on reboot"
|
||||
fi
|
||||
|
||||
local wcfg
|
||||
wcfg=$(ls "$HOST_ROOT"/mnt/c/Users/*/.wslconfig 2>/dev/null | head -1 || true)
|
||||
if [ -n "$wcfg" ] && grep -qE '^\s*memory\s*=' "$wcfg"; then
|
||||
echo " wslconfig memory set: $(grep -E '^\s*memory\s*=' "$wcfg" | tr -d ' ')"
|
||||
fact " wslconfig memory set: $(grep -E '^\s*memory\s*=' "$wcfg" | tr -d ' ')"
|
||||
else
|
||||
MANUAL+=("Cap/raise the WSL VM memory — see what is set versus what booted:
|
||||
make mem status
|
||||
make check mem
|
||||
It prints the edit to make and the command to apply it.")
|
||||
fi
|
||||
}
|
||||
|
||||
# Not a path check: /mnt is an ordinary mount point and an ext4 disk mounted
|
||||
# there is perfectly fine. What matters is the filesystem. The Windows drives
|
||||
# arrive as 9p (WSL2) or drvfs (WSL1); network and fuse mounts behave the same
|
||||
# way. None of them deliver inotify events, so anything watching files goes
|
||||
# quiet without saying why.
|
||||
# Filesystem types that deliver no inotify events (9p, drvfs, network, fuse).
|
||||
# Checks the fs type, not the path.
|
||||
watch_hostile_fs() {
|
||||
local dir="$1" fstype
|
||||
fstype=$(findmnt -no FSTYPE --target "$dir" 2>/dev/null || true)
|
||||
@@ -196,23 +252,73 @@ detect_filesystem() {
|
||||
on a $fstype mount, and everything else is slower:
|
||||
cp -r \"$root\" ~/ && cd ~/$(basename "$root")")
|
||||
else
|
||||
echo " filesystem $root ($(findmnt -no FSTYPE --target "$root" 2>/dev/null || echo local))"
|
||||
fact " filesystem $root ($(findmnt -no FSTYPE --target "$root" 2>/dev/null || echo local))"
|
||||
fi
|
||||
}
|
||||
|
||||
# tilt needs glibc >= 2.34 (measured on Amazon Linux 2). Report the version here;
|
||||
# `verify` catches the actual failure after installing.
|
||||
detect_libc() {
|
||||
local v=""
|
||||
if command -v ldd >/dev/null 2>&1; then
|
||||
v=$(ldd --version 2>/dev/null | head -1 | grep -oE '[0-9]+\.[0-9]+$' || true)
|
||||
fi
|
||||
if [ -z "$v" ]; then
|
||||
fact " libc unknown (no ldd) — 'verify' is the real test"
|
||||
return 0
|
||||
fi
|
||||
fact " libc glibc $v"
|
||||
if [ "$(printf '%s\n2.34\n' "$v" | sort -V | head -1)" != "2.34" ]; then
|
||||
echo " ! older than glibc 2.34, which tilt needs. kubectl, kind, jq and"
|
||||
echo " ctlptl are static or libc-only and work here; tilt will not start."
|
||||
echo " Install the core tier, or run tilt from a container."
|
||||
fi
|
||||
return 0
|
||||
}
|
||||
|
||||
# What this script itself needs, so `detect` answers "will install work?".
|
||||
detect_prereqs() {
|
||||
local missing=""
|
||||
if command -v curl >/dev/null 2>&1; then fact " download curl"
|
||||
elif command -v wget >/dev/null 2>&1; then fact " download wget"
|
||||
else echo " ! no curl and no wget — nothing can be downloaded"; missing+=" curl"
|
||||
fi
|
||||
|
||||
if command -v sha256sum >/dev/null 2>&1 || command -v shasum >/dev/null 2>&1; then
|
||||
fact " checksums ok"
|
||||
else
|
||||
echo " ! no sha256sum or shasum — downloads could not be verified"
|
||||
missing+=" coreutils"
|
||||
fi
|
||||
|
||||
if command -v tar >/dev/null 2>&1 && command -v gzip >/dev/null 2>&1; then
|
||||
fact " archives tar + gzip"
|
||||
else
|
||||
echo " ! no tar/gzip — tilt and ctlptl ship as tarballs, so the dev tier"
|
||||
echo " cannot be unpacked. The core tier is two bare binaries and is fine."
|
||||
missing+=" tar gzip"
|
||||
fi
|
||||
|
||||
if [ -n "$missing" ]; then
|
||||
MANUAL+=("Install what this script needs to run at all:
|
||||
$(pkg_install_cmd "${missing# }")")
|
||||
fi
|
||||
return 0
|
||||
}
|
||||
|
||||
detect_docker() {
|
||||
# Reachability of the daemon is the real question, and the CLI is only how
|
||||
# we ask it. Note that when this runs inside the installer container, Docker
|
||||
# necessarily exists on the host — otherwise nothing would be executing —
|
||||
# so a missing CLI in here is an installer packaging bug, not a host problem.
|
||||
# Daemon reachability is the real question; the CLI is only how we ask.
|
||||
if ! command -v docker >/dev/null 2>&1; then
|
||||
if [ -S /var/run/docker.sock ]; then
|
||||
echo " docker socket present (no cli in this context)"
|
||||
else
|
||||
echo " ! docker not found and no socket at /var/run/docker.sock"
|
||||
MANUAL+=("Install Docker — the one true prerequisite:
|
||||
sudo apt-get install -y docker.io && sudo usermod -aG docker \"\$USER\"
|
||||
then log out and back in.")
|
||||
MANUAL+=("Install Docker — the one true prerequisite, and the only thing here
|
||||
that needs root:
|
||||
$(pkg_install_cmd "$(docker_pkg)")
|
||||
sudo systemctl enable --now docker
|
||||
sudo usermod -aG docker \"\$USER\"
|
||||
then log out and back in, so the new group applies to your shell.")
|
||||
fi
|
||||
return
|
||||
fi
|
||||
@@ -220,12 +326,9 @@ detect_docker() {
|
||||
echo " docker $(docker version --format '{{.Server.Version}}' 2>/dev/null)"
|
||||
local n
|
||||
n=$(docker ps --filter "label=io.x-k8s.kind.cluster" --format '{{.Names}}' 2>/dev/null | wc -l)
|
||||
# Must be an `if`, not `[ ] && echo`: as the last statement in this
|
||||
# function the latter returns 1 when the count is zero, and `set -e`
|
||||
# then kills the caller. That is the fresh-machine case — no clusters
|
||||
# yet — so the bug only ever shows up where it does most harm.
|
||||
# Must be an `if`, not `[ ] && echo`: a zero count would return 1 under set -e.
|
||||
if [ "$n" -gt 0 ]; then
|
||||
echo " - $n kind node container(s) already running; see 'make cluster list'"
|
||||
echo " kind $n node container(s) running — 'make cluster list'"
|
||||
fi
|
||||
else
|
||||
echo " ! docker cli present but the daemon is unreachable"
|
||||
@@ -240,7 +343,7 @@ detect_inotify() {
|
||||
local w i
|
||||
w=$(cat /proc/sys/fs/inotify/max_user_watches 2>/dev/null || echo 0)
|
||||
i=$(cat /proc/sys/fs/inotify/max_user_instances 2>/dev/null || echo 0)
|
||||
echo " inotify watches=$w instances=$i"
|
||||
fact " inotify watches=$w instances=$i"
|
||||
|
||||
if [ "$w" -lt 524288 ] || [ "$i" -lt 512 ]; then
|
||||
echo " ! inotify limits are low — Tilt will silently stop noticing file changes"
|
||||
@@ -271,7 +374,7 @@ resolve_url() {
|
||||
|
||||
verify() {
|
||||
local file="$1" want="$2" name="$3" got
|
||||
got=$(sha256sum "$file" | awk '{print $1}')
|
||||
got=$($SHA "$file" | awk '{print $1}')
|
||||
if [ "$got" != "$want" ]; then
|
||||
echo "checksum mismatch for $name" >&2
|
||||
echo " expected $want" >&2
|
||||
@@ -285,7 +388,7 @@ fetch_bin() {
|
||||
local name="$1" url="$2" sha="$3" dest="$4"
|
||||
local tmp="$dest/.$name.tmp"
|
||||
echo " fetching $name"
|
||||
curl -fsSL --retry 3 -o "$tmp" "$(resolve_url "$url")"
|
||||
download "$(resolve_url "$url")" "$tmp"
|
||||
verify "$tmp" "$sha" "$name"
|
||||
mv "$tmp" "$dest/$name"
|
||||
chmod +x "$dest/$name"
|
||||
@@ -298,20 +401,15 @@ fetch_tgz() {
|
||||
local name="$1" url="$2" sha="$3" dest="$4" inner="$5" strip="$6"
|
||||
local tmp="$dest/.$name.tgz"
|
||||
echo " fetching $name"
|
||||
curl -fsSL --retry 3 -o "$tmp" "$(resolve_url "$url")"
|
||||
download "$(resolve_url "$url")" "$tmp"
|
||||
verify "$tmp" "$sha" "$name"
|
||||
# --no-same-owner: extracting as root would otherwise restore the uid/gid
|
||||
# baked into the archive (some ship as uid 1001), leaving a binary the host
|
||||
# user does not own.
|
||||
# --no-same-owner: as root, tar would restore the archive's uid/gid.
|
||||
tar -xzf "$tmp" -C "$dest" --strip-components="$strip" --no-same-owner "$inner"
|
||||
rm -f "$tmp"
|
||||
chmod +x "$dest/$name"
|
||||
}
|
||||
|
||||
# The installer runs as root so it can reach the docker socket, which means
|
||||
# everything it writes into a mounted volume lands root-owned and unusable from
|
||||
# the host. Hand it back to whoever owns the mount point (the host user created
|
||||
# that directory before mounting it).
|
||||
# The installer runs as root; hand files in a mounted dir back to the mount point's owner.
|
||||
fix_ownership() {
|
||||
local dir="$1"
|
||||
[ -d "$dir" ] || return 0
|
||||
@@ -323,33 +421,14 @@ fix_ownership() {
|
||||
chown -R "$owner" "$dir" 2>/dev/null || true
|
||||
}
|
||||
|
||||
# Two tiers, because not every machine should get cluster tooling.
|
||||
#
|
||||
# core kubectl, jq — talk to a cluster someone else runs. Nothing that
|
||||
# creates one. Appropriate on a managed or corporate-issued machine
|
||||
# where development tools are not wanted by default.
|
||||
# dev core plus kind and tilt — build clusters and hot-reload into them.
|
||||
#
|
||||
# The split exists because "install the toolchain" is not one decision: on a
|
||||
# managed workspace the right answer is kubectl and nothing else.
|
||||
# core: talk to a cluster someone else runs. dev: core plus tools that build clusters.
|
||||
CORE_TOOLS="kubectl jq"
|
||||
# No helm: every addon installs with `kubectl apply -f <url>`, so nothing here
|
||||
# has ever invoked it. Add it back the day something actually needs a chart.
|
||||
#
|
||||
# ctlptl is 'dev' rather than 'core' for the same reason kind is: core is "talk
|
||||
# to a cluster someone else runs", and ctlptl builds them. It earns its place
|
||||
# because it is what wires a cluster to a local registry — without one, an
|
||||
# unqualified image name resolves to docker.io/library/<name> and there is
|
||||
# nothing structural stopping a push there.
|
||||
DEV_TOOLS="kind tilt ctlptl"
|
||||
# No helm (nothing uses a chart). ctlptl wires in a local registry; compose is often
|
||||
# missing from distro docker packages.
|
||||
DEV_TOOLS="kind tilt ctlptl docker-compose"
|
||||
|
||||
# ── what is already on this machine ───────────────────────────────────────
|
||||
#
|
||||
# A tool already on PATH at its pinned version is left where it is. Without
|
||||
# this, install downloads a second copy into OUT_BIN and then reports the first
|
||||
# one as shadowed — noise, and wrong, when both are the same version. That is
|
||||
# the normal state of any machine someone set up by hand: the AWS Workspace
|
||||
# keeps its toolchain in ~/wdir/bin, all five at exactly these pins.
|
||||
# A tool already on PATH at its pinned version is left where it is.
|
||||
|
||||
pin_of() {
|
||||
case "$1" in
|
||||
@@ -358,12 +437,11 @@ pin_of() {
|
||||
kind) echo "$KIND_VERSION" ;;
|
||||
tilt) echo "$TILT_VERSION" ;;
|
||||
ctlptl) echo "$CTLPTL_VERSION" ;;
|
||||
docker-compose) echo "$COMPOSE_VERSION" ;;
|
||||
esac
|
||||
}
|
||||
|
||||
# The version string a binary reports. Each tool spells the question
|
||||
# differently, and kubectl has to be told --client or it goes looking for a
|
||||
# server to ask.
|
||||
# The version string a binary reports (kubectl needs --client).
|
||||
reported_version() {
|
||||
local tool="$1" path="$2"
|
||||
case "$tool" in
|
||||
@@ -373,13 +451,8 @@ reported_version() {
|
||||
esac
|
||||
}
|
||||
|
||||
# Does the binary at PATH report PIN? Matched as a whole version token, so
|
||||
# 0.37.6 never matches 10.37.60, with the leading v optional either side: kind
|
||||
# says v0.32.0, jq says jq-1.8.2, and tilt says v0.37.6 against a pin of 0.37.6.
|
||||
#
|
||||
# Bash's own regex rather than grep, deliberately. grep is not the same program
|
||||
# on every machine — some builds reject patterns that others accept — and a
|
||||
# failed grep inside a count reads exactly like a zero.
|
||||
# Does the binary at PATH report PIN? Whole-token match, leading v optional.
|
||||
# Bash regex rather than grep, deliberately.
|
||||
version_matches() {
|
||||
local tool="$1" path="$2" pin="$3" out v re
|
||||
out=$(reported_version "$tool" "$path") || return 1
|
||||
@@ -389,10 +462,8 @@ version_matches() {
|
||||
[[ $out =~ $re ]]
|
||||
}
|
||||
|
||||
# DEPS_ONLY narrows a fetch to the tools it names. Unset means the whole tier,
|
||||
# which is what an explicit `deps.sh fetch` always gets: "download these into
|
||||
# DIR" must not quietly skip something because this machine happens to have it.
|
||||
# Only install() sets it, to what detect_toolchain found missing or mismatched.
|
||||
# DEPS_ONLY narrows a fetch to the tools it names; unset means the whole tier.
|
||||
# Only install() sets it.
|
||||
want() { [ -z "${DEPS_ONLY:-}" ] || [[ " $DEPS_ONLY " == *" $1 "* ]]; }
|
||||
|
||||
# Every tool in the tier with its state, probed once and reported once. What
|
||||
@@ -401,16 +472,31 @@ TOOLCHAIN_NEED=""
|
||||
detect_toolchain() {
|
||||
local tier="${TIER:-dev}" b pin path found
|
||||
TOOLCHAIN_NEED=""
|
||||
local n=0
|
||||
echo
|
||||
echo "toolchain (pinned, tier '$tier')"
|
||||
fact "toolchain (pinned, tier '$tier')"
|
||||
for b in $(tier_tools "$tier"); do
|
||||
n=$((n + 1))
|
||||
pin=$(pin_of "$b")
|
||||
path=$(command -v "$b" 2>/dev/null || true)
|
||||
# compose is normally a docker CLI plugin, not on PATH: ask docker instead.
|
||||
if [ "$b" = docker-compose ] && [ -z "$path" ]; then
|
||||
if found=$(docker compose version --short 2>/dev/null) && [ -n "$found" ]; then
|
||||
if [ "${found#v}" = "${pin#v}" ]; then
|
||||
fact "$(printf " %-8s %-9s %s" "$b" "$pin" "docker cli plugin")"
|
||||
else
|
||||
printf " ! %-8s wants %s, the docker cli plugin reports '%s'\n" \
|
||||
"$b" "$pin" "$found"
|
||||
TOOLCHAIN_NEED+="$b "
|
||||
fi
|
||||
continue
|
||||
fi
|
||||
fi
|
||||
if [ -z "$path" ]; then
|
||||
printf " - %-8s %-9s not found\n" "$b" "$pin"
|
||||
TOOLCHAIN_NEED+="$b "
|
||||
elif version_matches "$b" "$path" "$pin"; then
|
||||
printf " %-8s %-9s %s\n" "$b" "$pin" "$path"
|
||||
fact "$(printf " %-8s %-9s %s" "$b" "$pin" "$path")"
|
||||
else
|
||||
found=$(reported_version "$b" "$path" 2>/dev/null | head -1 || true)
|
||||
printf " ! %-8s wants %s, %s reports '%s'\n" "$b" "$pin" "$path" "$found"
|
||||
@@ -418,9 +504,10 @@ detect_toolchain() {
|
||||
fi
|
||||
done
|
||||
if [ -z "$TOOLCHAIN_NEED" ]; then
|
||||
echo " every pinned tool is already on PATH — nothing to fetch"
|
||||
if [ -n "${VERBOSE:-}" ]; then echo " all $n on PATH — nothing to fetch"
|
||||
else echo "toolchain all $n pinned tools on PATH (tier $tier)"; fi
|
||||
else
|
||||
echo " 'make deps' fetches only: ${TOOLCHAIN_NEED% }"
|
||||
echo "toolchain 'make deps' fetches only: ${TOOLCHAIN_NEED% }"
|
||||
fi
|
||||
}
|
||||
|
||||
@@ -455,11 +542,13 @@ fetch() {
|
||||
if want kind; then fetch_bin kind "$KIND_URL" "$KIND_SHA256" "$dest"; fi
|
||||
if want tilt; then fetch_tgz tilt "$TILT_URL" "$TILT_SHA256" "$dest" tilt 0; fi
|
||||
if want ctlptl; then fetch_tgz ctlptl "$CTLPTL_URL" "$CTLPTL_SHA256" "$dest" ctlptl 0; fi
|
||||
if want docker-compose; then
|
||||
fetch_bin docker-compose "$COMPOSE_URL" "$COMPOSE_SHA256" "$dest"
|
||||
fi
|
||||
fi
|
||||
|
||||
fix_ownership "$dest"
|
||||
# kind writes the kubeconfig as root too; hand that back as well when it's
|
||||
# a mounted host directory rather than container-local state.
|
||||
# kind writes the kubeconfig as root too; hand that back as well.
|
||||
fix_ownership "${KUBE_DIR:-/out/kube}"
|
||||
}
|
||||
|
||||
@@ -481,10 +570,60 @@ report_manual() {
|
||||
done
|
||||
}
|
||||
|
||||
# Installing into a directory that sits early in PATH silently replaces whatever
|
||||
# the machine was already using — which on a shared or client machine can break
|
||||
# unrelated work (kubectl more than one minor away from a cluster is the common
|
||||
# one). Say so; never decide it for them.
|
||||
# A verified download proves the right file, not that this machine can run it
|
||||
# (old glibc breaks tilt). Run each one now.
|
||||
verify_tools() {
|
||||
local tier="${1:-dev}" b bin out rc broke=0
|
||||
echo "checking that each one actually runs"
|
||||
for b in $(tier_tools "$tier"); do
|
||||
bin="$OUT_BIN/$b"
|
||||
if [ ! -x "$bin" ]; then
|
||||
printf ' %-14s not installed\n' "$b"
|
||||
continue
|
||||
fi
|
||||
# Not piped into `head`: under pipefail, SIGPIPE (141) looked like failure.
|
||||
rc=0
|
||||
case "$b" in
|
||||
kubectl) out=$("$bin" version --client 2>&1) || rc=$? ;;
|
||||
jq) out=$("$bin" --version 2>&1) || rc=$? ;;
|
||||
*) out=$("$bin" version 2>&1) || rc=$? ;;
|
||||
esac
|
||||
out=${out%%$'\n'*}
|
||||
if [ "$rc" -eq 0 ]; then
|
||||
printf ' %-14s %s\n' "$b" "$out"
|
||||
else
|
||||
printf ' ! %-12s does not run here: %s\n' "$b" "$out"
|
||||
broke=1
|
||||
fi
|
||||
done
|
||||
if [ "$broke" -eq 1 ]; then
|
||||
echo
|
||||
echo " A binary that downloads and verifies but will not start is almost"
|
||||
echo " always this distro's libc being older than the release needs."
|
||||
echo " 'detect' prints the glibc version. The core tier (kubectl + jq)"
|
||||
echo " has no such dependency and will work regardless."
|
||||
fi
|
||||
return 0
|
||||
}
|
||||
|
||||
list() {
|
||||
echo "pinned, linux/amd64 only:"
|
||||
printf ' %-14s %s\n' kubectl "$KUBECTL_VERSION"
|
||||
printf ' %-14s %s\n' jq "$JQ_VERSION"
|
||||
printf ' %-14s %s\n' kind "$KIND_VERSION"
|
||||
printf ' %-14s %s\n' tilt "$TILT_VERSION"
|
||||
printf ' %-14s %s\n' ctlptl "$CTLPTL_VERSION"
|
||||
printf ' %-14s %s\n' docker-compose "$COMPOSE_VERSION"
|
||||
echo
|
||||
echo " core = $CORE_TOOLS"
|
||||
echo " dev = $CORE_TOOLS $DEV_TOOLS"
|
||||
echo
|
||||
echo "Checksums are pinned in the block at the top of this file. To bump one,"
|
||||
echo "take the new checksum from the publisher's own release list — the header"
|
||||
echo "comment has the exact commands."
|
||||
return 0
|
||||
}
|
||||
|
||||
tier_tools() { [ "$1" = "core" ] && echo "$CORE_TOOLS" || echo "$CORE_TOOLS $DEV_TOOLS"; }
|
||||
|
||||
warn_shadowing() {
|
||||
@@ -520,6 +659,24 @@ warn_shadowing() {
|
||||
OUT_BIN=\$PWD/def/bin make deps # then put that dir first in PATH")
|
||||
}
|
||||
|
||||
# Link the fetched docker-compose into ~/.docker/cli-plugins so `docker compose` works.
|
||||
install_compose_plugin() {
|
||||
local src="$OUT_BIN/docker-compose" dir="$HOME/.docker/cli-plugins"
|
||||
[ -x "$src" ] || return 0
|
||||
mkdir -p "$dir"
|
||||
# A real file there belongs to something else (docker-desktop, distro): don't overwrite.
|
||||
if [ -e "$dir/docker-compose" ] && [ ! -L "$dir/docker-compose" ]; then
|
||||
MANUAL+=("Something already installs the compose plugin at
|
||||
$dir/docker-compose
|
||||
To use rig's pinned build instead:
|
||||
ln -sf $src $dir/docker-compose")
|
||||
return 0
|
||||
fi
|
||||
ln -sfn "$src" "$dir/docker-compose"
|
||||
echo " compose plugin -> $dir/docker-compose"
|
||||
return 0
|
||||
}
|
||||
|
||||
install() {
|
||||
local tier="${1:-dev}" b
|
||||
TIER="$tier"
|
||||
@@ -538,10 +695,12 @@ install() {
|
||||
if [ "$tier" = "core" ]; then
|
||||
echo " (no kind/tilt — 'make deps dev' adds them)"
|
||||
fi
|
||||
# Only when compose was fetched, never at a copy rig did not install.
|
||||
case " $TOOLCHAIN_NEED " in
|
||||
*" docker-compose "*) install_compose_plugin ;;
|
||||
esac
|
||||
|
||||
# Only worth saying when something actually landed in OUT_BIN. When every
|
||||
# tool was satisfied elsewhere, OUT_BIN may reasonably be off PATH, and
|
||||
# telling the user to add it would be advice to fix nothing.
|
||||
# PATH advice only when something actually landed in OUT_BIN.
|
||||
case ":${PATH}:" in
|
||||
*":$OUT_BIN:"*) ;;
|
||||
*) MANUAL+=("Put the toolchain on your PATH — add to ~/.bashrc:
|
||||
@@ -557,9 +716,75 @@ install() {
|
||||
|
||||
require_linux
|
||||
|
||||
case "${1:-install}" in
|
||||
detect) detect; report_manual ;;
|
||||
fetch) shift; fetch "$@" ;;
|
||||
install) shift; install "${1:-dev}" ;;
|
||||
*) echo "usage: $0 [detect|fetch|install]" >&2; exit 1 ;;
|
||||
# Shift only if there is an argument: a bare `shift` returns 1 under set -e.
|
||||
cmd="${1:-install}"
|
||||
[ $# -gt 0 ] && shift
|
||||
|
||||
# ── manifests rig's own addons apply ───────────────────────────────────────
|
||||
# Pinned (URL + SHA256), fetched through the same DEPS_SOURCE resolver as the
|
||||
# binaries and verified, then applied from disk: an offline machine needs no
|
||||
# network for them. Default home: vendor/manifests/ in rig's folder (gitignored).
|
||||
MANIFESTS_HOME="${MANIFESTS_HOME:-$(cd .. && pwd)/vendor/manifests}"
|
||||
BAKED_MANIFESTS="${BAKED_MANIFESTS:-/opt/rig/manifests}"
|
||||
MANIFEST_NAMES="METALLB CERT_MANAGER METRICS_SERVER"
|
||||
|
||||
manifest_path() { # NAME dir
|
||||
local v="${1}_VERSION"
|
||||
echo "$2/$(echo "$1" | tr 'A-Z_' 'a-z-')-${!v}.yaml"
|
||||
}
|
||||
|
||||
# Make one pinned manifest present and verified in dir; print only its path.
|
||||
fetch_manifest() { # NAME dir
|
||||
local name="$1" dir="$2" url_var="${1}_MANIFEST_URL" sha_var="${1}_MANIFEST_SHA256" file
|
||||
if [ -z "${!url_var:-}" ] || [ -z "${!sha_var:-}" ]; then
|
||||
echo "no pinned manifest for $name (${url_var} / ${sha_var} unset)" >&2
|
||||
exit 1
|
||||
fi
|
||||
file=$(manifest_path "$name" "$dir")
|
||||
if [ -f "$file" ] && [ "$($SHA "$file" | awk '{print $1}')" = "${!sha_var}" ]; then
|
||||
echo "$file"
|
||||
return
|
||||
fi
|
||||
mkdir -p "$dir"
|
||||
if [ "$DEPS_SOURCE" = baked ]; then
|
||||
cp "$(manifest_path "$name" "$BAKED_MANIFESTS")" "$file.tmp"
|
||||
else
|
||||
download "$(resolve_url "${!url_var}")" "$file.tmp"
|
||||
fi
|
||||
verify "$file.tmp" "${!sha_var}" "$name manifest"
|
||||
mv "$file.tmp" "$file"
|
||||
echo "$file"
|
||||
}
|
||||
|
||||
fetch_manifests() { # [--to DIR]
|
||||
local dest="$MANIFESTS_HOME" n
|
||||
if [ "${1:-}" = --to ]; then dest="$(abspath "${2:?--to needs a directory}")"; fi
|
||||
echo "fetching the addons' manifests into $dest (source: $DEPS_SOURCE)"
|
||||
for n in $MANIFEST_NAMES; do
|
||||
echo " $n $(fetch_manifest "$n" "$dest")"
|
||||
done
|
||||
}
|
||||
|
||||
# Baked mode copies binaries already in the image, so it needs no downloader.
|
||||
need_downloads() {
|
||||
require_amd64
|
||||
if [ "$DEPS_SOURCE" != baked ]; then pick_downloader; fi
|
||||
pick_sha
|
||||
}
|
||||
|
||||
case "$cmd" in
|
||||
detect) if [ "${1:-}" = all ]; then VERBOSE=1; fi; detect; report_manual ;;
|
||||
list) list ;;
|
||||
verify) verify_tools "${1:-dev}" ;;
|
||||
fetch) need_downloads; fetch "$@" ;;
|
||||
install) need_downloads; install "${1:-dev}" ;;
|
||||
manifest) need_downloads
|
||||
fetch_manifest "${1:?usage: $0 manifest <METALLB|CERT_MANAGER|METRICS_SERVER>}" "$MANIFESTS_HOME" ;;
|
||||
manifests) need_downloads; fetch_manifests "$@" ;;
|
||||
*) echo "usage: $0 [detect [all]|list|verify|fetch|install|manifest NAME|manifests]" >&2
|
||||
echo " install [core|dev] (default dev)" >&2
|
||||
echo " fetch [core|dev] [--to DIR]" >&2
|
||||
echo " manifests [--to DIR] the addons' pinned manifests, verified" >&2
|
||||
echo " OUT_BIN=<dir> overrides the install directory" >&2
|
||||
exit 1 ;;
|
||||
esac
|
||||
|
||||
@@ -1,272 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Share ONE Docker daemon across WSL distros, instead of running one per distro.
|
||||
#
|
||||
# Why this exists
|
||||
# ---------------
|
||||
# WSL2 distros share a kernel and a network stack. Two dockerd instances then
|
||||
# contend over docker0 and iptables, which can disturb the daemon you actually
|
||||
# depend on. Docker Desktop avoids this by running a single daemon in a
|
||||
# dedicated distro and sharing its socket — this is the same idea, without
|
||||
# Docker Desktop.
|
||||
#
|
||||
# So a throwaway rig box does NOT install Docker. It borrows the daemon from
|
||||
# whichever distro is the designated host. That also makes the test more honest:
|
||||
# rig never installs Docker anyway — Docker is its documented prerequisite.
|
||||
#
|
||||
# How
|
||||
# ---
|
||||
# /mnt/wsl is a tmpfs with `shared` mount propagation, visible to every distro
|
||||
# in the WSL VM. The owning distro exposes its socket there; guests point
|
||||
# DOCKER_HOST at it. Two ways, with different costs:
|
||||
#
|
||||
# share bind-mount the existing socket onto the shared tmpfs.
|
||||
# Instant, and dockerd is NEVER restarted. Lasts until the
|
||||
# next WSL shutdown.
|
||||
# share --persist additionally install a systemd drop-in so dockerd listens
|
||||
# there itself. Survives restarts, but requires one Docker
|
||||
# restart now — which stops every container that has no
|
||||
# restart policy, since live-restore is off by default.
|
||||
#
|
||||
# The bind mount is the default precisely because the persistent version's cost
|
||||
# is paid on a machine that is already working.
|
||||
#
|
||||
# Reversibility is the whole design
|
||||
# ---------------------------------
|
||||
# `unshare` removes the bind mount (no restart) and, if present, the drop-in.
|
||||
# The original systemd unit is never edited — only an additive drop-in file is
|
||||
# ever created — so undoing is deletion, not repair. `status` always states
|
||||
# which of the three roles a distro is in, in those words.
|
||||
#
|
||||
# Nothing here runs automatically. It does nothing until invoked.
|
||||
#
|
||||
# Usage:
|
||||
# dockerhost.sh status # which distro owns Docker; what this one uses
|
||||
# dockerhost.sh share # share it (bind mount, no daemon restart)
|
||||
# dockerhost.sh share --persist # ...and survive WSL restarts (restarts Docker)
|
||||
# dockerhost.sh unshare # undo it; this distro owns its Docker again
|
||||
# dockerhost.sh use [--persist] # point THIS distro at the shared socket
|
||||
set -euo pipefail
|
||||
|
||||
SHARED_DIR=/mnt/wsl/shared-docker
|
||||
SHARED_SOCK="$SHARED_DIR/docker.sock"
|
||||
OWNER_FILE="$SHARED_DIR/OWNER"
|
||||
DROPIN=/etc/systemd/system/docker.service.d/10-rig-shared-socket.conf
|
||||
PROFILE_D=/etc/profile.d/rig-docker-host.sh
|
||||
|
||||
distro_name() { echo "${WSL_DISTRO_NAME:-$(hostname)}"; }
|
||||
|
||||
require_wsl() {
|
||||
grep -qi microsoft /proc/version 2>/dev/null && return 0
|
||||
echo "dockerhost is WSL-only: it relies on /mnt/wsl being shared between distros." >&2
|
||||
exit 1
|
||||
}
|
||||
|
||||
# ── status ─────────────────────────────────────────────────────────────────
|
||||
|
||||
status() {
|
||||
require_wsl
|
||||
echo "distro $(distro_name)"
|
||||
|
||||
if [ -f "$DROPIN" ] || mountpoint -q "$SHARED_SOCK" 2>/dev/null; then
|
||||
echo "role SHARING — this distro's Docker is offered to other distros"
|
||||
elif [ -n "${DOCKER_HOST:-}" ] && [ "${DOCKER_HOST}" = "unix://$SHARED_SOCK" ]; then
|
||||
echo "role BORROWING — using another distro's Docker"
|
||||
else
|
||||
echo "role standalone — this WSL installation has the main host Docker"
|
||||
fi
|
||||
|
||||
echo
|
||||
if [ -S "$SHARED_SOCK" ]; then
|
||||
echo "shared sock $SHARED_SOCK (present)"
|
||||
[ -f "$OWNER_FILE" ] && sed 's/^/ /' "$OWNER_FILE"
|
||||
else
|
||||
echo "shared sock none — no distro is sharing right now"
|
||||
fi
|
||||
|
||||
echo
|
||||
echo "DOCKER_HOST ${DOCKER_HOST:-(unset — using /var/run/docker.sock)}"
|
||||
if command -v docker >/dev/null 2>&1; then
|
||||
echo "docker $(docker version --format '{{.Server.Version}}' 2>/dev/null || echo unreachable)"
|
||||
else
|
||||
echo "docker cli not installed"
|
||||
fi
|
||||
}
|
||||
|
||||
# ── share / unshare (run on the host distro) ───────────────────────────────
|
||||
|
||||
# Default: expose the EXISTING socket by bind-mounting it onto the shared tmpfs.
|
||||
# /mnt/wsl has `shared` propagation, so the mount is visible in other distros.
|
||||
#
|
||||
# The point of doing it this way is that dockerd is never restarted. Restarting
|
||||
# it stops every container that has no restart policy (live-restore is off by
|
||||
# default), which on a working machine means quietly killing whatever you had
|
||||
# running. Not a trade worth making just to expose a socket.
|
||||
#
|
||||
# Cost: a bind mount does not survive a WSL VM shutdown. `--persist` adds the
|
||||
# systemd drop-in as well, which does survive but needs that one restart.
|
||||
share_bind() {
|
||||
mkdir -p "$SHARED_DIR"
|
||||
chmod 0755 "$SHARED_DIR"
|
||||
|
||||
if mountpoint -q "$SHARED_SOCK" 2>/dev/null; then
|
||||
echo "already bind-mounted at $SHARED_SOCK"
|
||||
else
|
||||
[ -S /var/run/docker.sock ] || { echo "no /var/run/docker.sock here" >&2; exit 1; }
|
||||
# The target must exist as a file for a bind mount onto it.
|
||||
[ -e "$SHARED_SOCK" ] || : > "$SHARED_SOCK"
|
||||
mount --bind /var/run/docker.sock "$SHARED_SOCK"
|
||||
echo "bind-mounted /var/run/docker.sock -> $SHARED_SOCK (no daemon restart)"
|
||||
fi
|
||||
|
||||
cat > "$OWNER_FILE" <<EOF
|
||||
owner distro: $(distro_name)
|
||||
docker gid: $(getent group docker | cut -d: -f3)
|
||||
socket: $SHARED_SOCK
|
||||
method: bind-mount (until the next WSL shutdown)
|
||||
EOF
|
||||
}
|
||||
|
||||
share() {
|
||||
require_wsl
|
||||
[ "$(id -u)" -eq 0 ] || { echo "run with sudo: sudo bash ctrl/dockerhost.sh share" >&2; exit 1; }
|
||||
|
||||
share_bind
|
||||
|
||||
if [ "${1:-}" != "--persist" ]; then
|
||||
echo
|
||||
echo "This lasts until the next WSL shutdown. To make it survive, re-run with"
|
||||
echo "--persist — but note that adds a systemd drop-in and RESTARTS Docker,"
|
||||
echo "which stops any container that has no restart policy."
|
||||
return 0
|
||||
fi
|
||||
|
||||
if [ -f "$DROPIN" ]; then
|
||||
echo "drop-in already present — sharing persists across restarts."
|
||||
return 0
|
||||
fi
|
||||
|
||||
echo
|
||||
echo "--persist: installing a systemd drop-in and restarting Docker."
|
||||
echo "Containers without a restart policy will stop and will NOT come back."
|
||||
docker ps --format ' {{.Names}} restart={{.HostConfig.RestartPolicy.Name}}' 2>/dev/null \
|
||||
|| docker ps --format ' {{.Names}}' 2>/dev/null || true
|
||||
echo
|
||||
|
||||
local exec_line
|
||||
exec_line=$(systemctl cat docker.service | grep -m1 '^ExecStart=')
|
||||
if [ -z "$exec_line" ]; then
|
||||
echo "could not read docker.service ExecStart — refusing to guess" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
mkdir -p "$(dirname "$DROPIN")" "$SHARED_DIR"
|
||||
# Additive only: blank the inherited ExecStart, then restate it verbatim
|
||||
# with one extra -H. Nothing about the original unit is edited.
|
||||
cat > "$DROPIN" <<EOF
|
||||
# Added by rig (ctrl/dockerhost.sh share).
|
||||
#
|
||||
# Adds a SECOND listening socket on the WSL-shared tmpfs so other distros can
|
||||
# use this daemon instead of running their own. The original socket is
|
||||
# untouched, so this distro behaves exactly as before.
|
||||
#
|
||||
# To undo: sudo bash ctrl/dockerhost.sh unshare
|
||||
[Service]
|
||||
ExecStartPre=-/bin/mkdir -p $SHARED_DIR
|
||||
ExecStartPre=-/bin/chmod 0755 $SHARED_DIR
|
||||
ExecStart=
|
||||
${exec_line} -H unix://$SHARED_SOCK
|
||||
EOF
|
||||
|
||||
systemctl daemon-reload
|
||||
systemctl restart docker
|
||||
|
||||
# Guests need a group with a MATCHING GID to use the socket; GIDs are not
|
||||
# consistent across distros, so record ours rather than assume.
|
||||
cat > "$OWNER_FILE" <<EOF
|
||||
owner distro: $(distro_name)
|
||||
docker gid: $(getent group docker | cut -d: -f3)
|
||||
socket: $SHARED_SOCK
|
||||
EOF
|
||||
|
||||
echo "sharing from '$(distro_name)'"
|
||||
echo " guests: export DOCKER_HOST=unix://$SHARED_SOCK"
|
||||
echo " undo: sudo bash ctrl/dockerhost.sh unshare"
|
||||
echo
|
||||
echo "NOTE: /mnt/wsl is tmpfs and is cleared when the WSL VM shuts down."
|
||||
echo " The drop-in recreates the directory on the next Docker start."
|
||||
}
|
||||
|
||||
unshare_() {
|
||||
require_wsl
|
||||
[ "$(id -u)" -eq 0 ] || { echo "run with sudo: sudo bash ctrl/dockerhost.sh unshare" >&2; exit 1; }
|
||||
|
||||
local did=0
|
||||
|
||||
# The bind mount first: undoing it needs no restart, so a plain `share`
|
||||
# is fully reversible without disturbing anything.
|
||||
if mountpoint -q "$SHARED_SOCK" 2>/dev/null; then
|
||||
umount "$SHARED_SOCK"
|
||||
rm -f "$SHARED_SOCK"
|
||||
echo " removed the bind mount (no restart needed)"
|
||||
did=1
|
||||
fi
|
||||
rm -f "$OWNER_FILE"
|
||||
rmdir "$SHARED_DIR" 2>/dev/null || true
|
||||
|
||||
if [ -f "$DROPIN" ]; then
|
||||
rm -f "$DROPIN"
|
||||
rmdir "$(dirname "$DROPIN")" 2>/dev/null || true
|
||||
systemctl daemon-reload
|
||||
systemctl restart docker
|
||||
echo " removed the systemd drop-in and restarted Docker"
|
||||
did=1
|
||||
fi
|
||||
|
||||
if [ "$did" -eq 0 ]; then
|
||||
echo "not sharing — this WSL installation already has the main host Docker."
|
||||
return 0
|
||||
fi
|
||||
|
||||
echo "restored: this WSL installation has the main host Docker again."
|
||||
echo " (nothing else was changed; the original unit was never edited)"
|
||||
}
|
||||
|
||||
# ── use (run on a guest distro) ────────────────────────────────────────────
|
||||
|
||||
use() {
|
||||
require_wsl
|
||||
if [ ! -S "$SHARED_SOCK" ]; then
|
||||
echo "no shared socket at $SHARED_SOCK" >&2
|
||||
echo "Run 'sudo bash ctrl/dockerhost.sh share' in the distro that owns Docker." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Align the local docker group GID with the owner's, or the socket is
|
||||
# unreadable here even though it is visible.
|
||||
if [ -f "$OWNER_FILE" ] && [ "$(id -u)" -eq 0 ]; then
|
||||
local gid; gid=$(awk '/docker gid:/ {print $3}' "$OWNER_FILE")
|
||||
if [ -n "$gid" ]; then
|
||||
if getent group docker >/dev/null; then
|
||||
[ "$(getent group docker | cut -d: -f3)" = "$gid" ] || groupmod -g "$gid" docker
|
||||
else
|
||||
groupadd -g "$gid" docker
|
||||
fi
|
||||
fi
|
||||
fi
|
||||
|
||||
if [ "${1:-}" = "--persist" ]; then
|
||||
[ "$(id -u)" -eq 0 ] || { echo "--persist needs root" >&2; exit 1; }
|
||||
echo "export DOCKER_HOST=unix://$SHARED_SOCK" > "$PROFILE_D"
|
||||
echo "persisted in $PROFILE_D"
|
||||
fi
|
||||
|
||||
echo "export DOCKER_HOST=unix://$SHARED_SOCK"
|
||||
}
|
||||
|
||||
case "${1:-status}" in
|
||||
status) status ;;
|
||||
share) shift; share "${1:-}" ;;
|
||||
unshare) unshare_ ;;
|
||||
use) shift; use "${1:-}" ;;
|
||||
*) echo "usage: $0 [status|share|unshare|use [--persist]]" >&2; exit 1 ;;
|
||||
esac
|
||||
@@ -1,16 +1,7 @@
|
||||
#!/usr/bin/env bash
|
||||
# Documentation: render the diagrams, and serve the pages.
|
||||
#
|
||||
# The docs are the instructions for building the cluster, so they must work
|
||||
# BEFORE anything else exists. That rules out serving them from the cluster, and
|
||||
# it rules out python -m http.server too — a minimal Debian has no python3. What
|
||||
# it does have, by definition, is Docker: the single prerequisite rig already
|
||||
# demands. So a throwaway nginx container serves a read-only bind mount.
|
||||
#
|
||||
# Rendered SVGs are committed alongside their .dot sources for the same reason:
|
||||
# the pages have to read on a machine with no Graphviz installed.
|
||||
#
|
||||
# Documentation: render the diagrams, and serve the pages from a throwaway nginx container.
|
||||
# Usage: docs.sh serve | graphs
|
||||
# Notes: docs/notes/docs.md
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")"
|
||||
|
||||
|
||||
@@ -1,30 +0,0 @@
|
||||
# client — the regulated-estate shape. Multi-node so taints, affinity and
|
||||
# topology are real; apiserver audit on; images through a pull-through cache of
|
||||
# the corporate registry.
|
||||
#
|
||||
# Costs roughly 4-6 GB. Check `make cluster list` before starting this alongside
|
||||
# other work — see the memory note in the README.
|
||||
|
||||
PROFILE_NAME=client
|
||||
K8S_VERSION=v1_36
|
||||
KIND_CONFIG=kind-config.client.yaml.tpl
|
||||
ADDONS="metallb cert-manager metrics-server"
|
||||
REGISTRY_MODE=mirror
|
||||
INGRESS_MODE=hostport
|
||||
DNS_MODE=hosts
|
||||
|
||||
# Ports derive from the directory name by default (see ctrl/ports.sh), so
|
||||
# several environments run side by side.
|
||||
#
|
||||
# Opt in to the real ports below only when this is the ONLY environment and
|
||||
# nothing else owns :80. They fail to bind otherwise, and docker reports it as an
|
||||
# opaque "failed to bind host port 0.0.0.0:80/tcp: address already in use"
|
||||
# halfway through cluster creation. `make check` checks before you spend the
|
||||
# time. Uncommenting also means only one environment can exist at a time.
|
||||
# HTTP_PORT=80
|
||||
# HTTPS_PORT=443
|
||||
|
||||
# Set these in ctrl/.env (gitignored), not here:
|
||||
# REGISTRY_REMOTE_URL=https://artifactory.corp.example/artifactory/api/docker/docker-virtual
|
||||
# REGISTRY_USER / REGISTRY_PASSWORD
|
||||
# REGISTRY_CA_FILE=/path/to/corp-root-ca.crt
|
||||
@@ -1,41 +0,0 @@
|
||||
# data — a cluster with the cabinets an environment asks for.
|
||||
#
|
||||
# A cabinet is a public service dropped in as-is — the upstream image,
|
||||
# unmodified, reachable at a known address. It is declared once and installs on
|
||||
# either target: a `service.yml` composes it for a laptop, and the addons below
|
||||
# install the same one here. The names match deliberately — each cabinet.json
|
||||
# carries a `rig_addon` field pointing at ctrl/addons/<name>.sh.
|
||||
#
|
||||
# Everything lands in the `data` namespace (DATA_NAMESPACE to move it), so
|
||||
# `make cluster reset` on the app namespace leaves the databases alone.
|
||||
#
|
||||
# Costs roughly 2-3 GB with airflow, under 1 without. Airflow's first boot runs
|
||||
# the whole metadata migration, so expect a few minutes before it is ready.
|
||||
|
||||
PROFILE_NAME=data
|
||||
K8S_VERSION=v1_36
|
||||
KIND_CONFIG=kind-config.yaml.tpl
|
||||
# Order matters: addons.sh installs in the order listed, and airflow refuses to
|
||||
# start without a metadata database, so postgres comes first.
|
||||
ADDONS="metallb postgres redis airflow"
|
||||
# local, not none — see minimal.env: `none` has no outward-push guard.
|
||||
REGISTRY_MODE=local
|
||||
INGRESS_MODE=hostport
|
||||
DNS_MODE=hosts
|
||||
|
||||
# Namespace for the dependency containers.
|
||||
DATA_NAMESPACE=data
|
||||
|
||||
# Postgres identity. The password is not here: postgres.sh generates one on
|
||||
# first install and keeps it across re-runs, so re-running the addon never
|
||||
# rotates the credential out from under whatever is already connected.
|
||||
POSTGRES_DB=app
|
||||
POSTGRES_USER=app
|
||||
POSTGRES_STORAGE=2Gi
|
||||
|
||||
AIRFLOW_ADMIN_USER=admin
|
||||
|
||||
# Ports derive from the directory name by default — see ctrl/ports.sh. Reach
|
||||
# the databases with port-forward rather than binding more host ports:
|
||||
# kubectl -n data port-forward svc/postgres 5432:5432
|
||||
# kubectl -n data port-forward svc/airflow 8080:8080
|
||||
@@ -1,21 +0,0 @@
|
||||
# minimal — the default. One node, no addons, no registry.
|
||||
# Assumes nothing and boots fast. Start here; move to client.env when you need
|
||||
# the regulated behaviours.
|
||||
#
|
||||
|
||||
PROFILE_NAME=minimal
|
||||
K8S_VERSION=v1_36
|
||||
KIND_CONFIG=kind-config.yaml.tpl
|
||||
ADDONS=""
|
||||
# local, not none: `none` leaves the cluster with no registry to push to, and an
|
||||
# unqualified image name then means docker.io/library/<name>. In a regulated
|
||||
# estate that is a disclosure risk, not a convenience trade — so the default
|
||||
# carries the guard even though it costs one container.
|
||||
REGISTRY_MODE=local
|
||||
INGRESS_MODE=hostport
|
||||
DNS_MODE=hosts
|
||||
|
||||
# Ports are deliberately NOT set here. They derive from the directory name so
|
||||
# several environments coexist — see ctrl/ports.sh, and `make ports` to see the
|
||||
# block this one gets. A fixed default here would collide with whatever else the
|
||||
# machine happens to be running; 8080 in particular is rarely free.
|
||||
21
rig/ctrl/env.d/mirror.env.example
Normal file
21
rig/ctrl/env.d/mirror.env.example
Normal file
@@ -0,0 +1,21 @@
|
||||
# EXAMPLE PROFILE (optional): copy to mirror.env, then PROFILE=mirror; overlays the defaults.
|
||||
# mirror — images via a pull-through cache of an internal registry, TLS and metrics addons.
|
||||
# A profile says how this machine reaches the world; what runs is an overlay's business.
|
||||
# Notes: docs/notes/env.md
|
||||
|
||||
PROFILE_NAME=mirror
|
||||
K8S_VERSION=v1_36
|
||||
ADDONS="metallb cert-manager metrics-server"
|
||||
REGISTRY_MODE=mirror
|
||||
INGRESS_MODE=hostport
|
||||
DNS_MODE=hosts
|
||||
|
||||
# Ports derive from the directory name by default (see ctrl/ports.sh).
|
||||
# Real ports only if this is the ONLY environment and nothing owns :80; `make check` tests it.
|
||||
# HTTP_PORT=80
|
||||
# HTTPS_PORT=443
|
||||
|
||||
# Set these in ctrl/.env (gitignored), not here:
|
||||
# REGISTRY_REMOTE_URL=https://registry.internal.example/api/docker/docker-virtual
|
||||
# REGISTRY_USER / REGISTRY_PASSWORD
|
||||
# REGISTRY_CA_FILE=/path/to/internal-root-ca.crt
|
||||
@@ -1,18 +0,0 @@
|
||||
# offline — air-gapped. Everything comes from a local registry that was loaded
|
||||
# ahead of time; nothing reaches the internet. Pair with the deps-full image
|
||||
# (DEPS_SOURCE=baked) so the toolchain install is offline too.
|
||||
#
|
||||
# The heavier addons are left out to keep first boot viable.
|
||||
|
||||
PROFILE_NAME=offline
|
||||
K8S_VERSION=v1_36
|
||||
KIND_CONFIG=kind-config.audit.yaml.tpl
|
||||
ADDONS="metallb"
|
||||
REGISTRY_MODE=local
|
||||
INGRESS_MODE=hostport
|
||||
DNS_MODE=hosts
|
||||
|
||||
# Derived from the directory name by default — see ctrl/ports.sh.
|
||||
# Uncomment for the real ports, but only if this is the only environment.
|
||||
# HTTP_PORT=80
|
||||
# HTTPS_PORT=443
|
||||
15
rig/ctrl/env.d/offline.env.example
Normal file
15
rig/ctrl/env.d/offline.env.example
Normal file
@@ -0,0 +1,15 @@
|
||||
# EXAMPLE PROFILE (optional): copy to offline.env, then PROFILE=offline; overlays the defaults.
|
||||
# offline — air-gapped: images from a preloaded local registry; pair with DEPS_SOURCE=baked.
|
||||
# Notes: docs/notes/env.md
|
||||
|
||||
PROFILE_NAME=offline
|
||||
K8S_VERSION=v1_36
|
||||
ADDONS="metallb"
|
||||
REGISTRY_MODE=local
|
||||
INGRESS_MODE=hostport
|
||||
DNS_MODE=hosts
|
||||
|
||||
# Derived from the directory name by default — see ctrl/ports.sh.
|
||||
# Uncomment for the real ports, but only if this is the only environment.
|
||||
# HTTP_PORT=80
|
||||
# HTTPS_PORT=443
|
||||
@@ -1,15 +0,0 @@
|
||||
# /etc/hosts block for this environment. Rendered by newbox.sh; ${CLUSTER} and
|
||||
# ${HTTP_PORT} are substituted.
|
||||
#
|
||||
# Hostnames are a convenience, not a requirement — every service is reachable at
|
||||
# localhost:<port> without any of this, which is why DNS is not touched by
|
||||
# default. Add entries here as the model grows.
|
||||
#
|
||||
# On Windows the same block has to go in
|
||||
# C:\Windows\System32\drivers\etc\hosts for a browser to resolve these. That
|
||||
# file does NOT support wildcards, so every name must be listed explicitly.
|
||||
# newbox.sh prints the block for you to paste rather than editing it.
|
||||
|
||||
127.0.0.1 ${CLUSTER}.local
|
||||
127.0.0.1 api.${CLUSTER}.local
|
||||
127.0.0.1 docs.${CLUSTER}.local
|
||||
@@ -1,70 +0,0 @@
|
||||
# `ctrl/k8s` — cluster shape, and what runs on it
|
||||
|
||||
Same layout as every other project here: a kind config, a kustomize `base/`,
|
||||
and an `overlays/dev/` that patches it.
|
||||
|
||||
```
|
||||
kind-config*.yaml.tpl the cluster itself — nodes, ports, audit
|
||||
base/ the components, as plain manifests
|
||||
overlays/dev/ how this rig differs from the base
|
||||
audit-policy.yaml mounted into the apiserver by the audit shapes
|
||||
```
|
||||
|
||||
## Why the cluster config is a template
|
||||
|
||||
Every other project checks in a literal `kind-config.yaml`, because there is
|
||||
exactly one `unt` and one `nvi`. A rig is copied and renamed to make a second
|
||||
environment, and both the cluster name and the host port block follow the
|
||||
directory name — so a literal would make every copy collide on both.
|
||||
|
||||
`ctrl/cluster.sh` renders it with `sed`, substituting `${CLUSTER}`,
|
||||
`${NODE_IMAGE}`, `${HTTP_PORT}` and `${HOST_WORKDIR}`. Not `envsubst`: that is
|
||||
`gettext-base`, which a minimal Debian does not have, and Docker being the only
|
||||
prerequisite is the one promise rig makes.
|
||||
|
||||
**The chosen file is the source of truth for node count and audit.**
|
||||
`lib/config.sh` reads both back out of it, so a profile names a shape and does
|
||||
not restate what the YAML already says.
|
||||
|
||||
| file | nodes | audit | profiles |
|
||||
| --- | --- | --- | --- |
|
||||
| `kind-config.yaml.tpl` | 1 | off | `minimal`, `data` |
|
||||
| `kind-config.audit.yaml.tpl` | 1 | on | `offline` |
|
||||
| `kind-config.client.yaml.tpl` | 3 | on | `client` |
|
||||
|
||||
A profile picks one with `KIND_CONFIG` in `ctrl/env.d/<profile>.env`. Adding a
|
||||
shape is adding a file — there is no dispatcher to edit.
|
||||
|
||||
Audit is an apiserver flag and therefore fixed at creation: changing it is
|
||||
`make cluster reset`, not a re-apply.
|
||||
|
||||
## `base/` — replace these
|
||||
|
||||
**The two components in `base/` are examples, not the system.** They exist so
|
||||
the real manifests have a shape to be written against.
|
||||
|
||||
The real ones are expected to be versioned **separately from the installer** —
|
||||
they change on a different cadence, by different people, under different review.
|
||||
Point `MANIFESTS_DIR` in `ctrl/.env` at their overlay and rig stops owning them:
|
||||
|
||||
```
|
||||
MANIFESTS_DIR=../platform-manifests/overlays/dev
|
||||
```
|
||||
|
||||
Until then it defaults to `ctrl/k8s/overlays/dev`.
|
||||
|
||||
### The three states a component can be in
|
||||
|
||||
Switching between them should be a one-line change, never a rewrite. The DNS
|
||||
name stays the same in every case, so callers never know the difference:
|
||||
|
||||
| state | what exists | when |
|
||||
| --- | --- | --- |
|
||||
| **real** | an image built from source, hot-reloaded | the one thing you are working on |
|
||||
| **mock** | a stub returning canned responses (`example-mock.yaml`) | everything else — most of the estate |
|
||||
| **remote** | no pod at all, just a Service (`example-remote.yaml`) | when the real system is reachable and you want it |
|
||||
|
||||
Most components should be **mock**. What has to be faithful is the topology —
|
||||
names, ports, dependency order, who can reach whom, how it fails. The workloads
|
||||
are noise, and mocking them is what makes several copies of a large estate fit
|
||||
on one laptop.
|
||||
@@ -1,44 +0,0 @@
|
||||
# Apiserver audit policy. Mounted into the control plane at creation when a
|
||||
# profile sets AUDIT=on — an apiserver flag, so it cannot be added to a running
|
||||
# cluster without recreating it.
|
||||
#
|
||||
# Deliberately modest: enough to make "who changed what, and when" answerable
|
||||
# during onboarding without filling the disk. Read the log with:
|
||||
# docker exec <cluster>-control-plane cat /var/log/kubernetes/audit.log
|
||||
apiVersion: audit.k8s.io/v1
|
||||
kind: Policy
|
||||
|
||||
# Never log the request body for these — they contain credentials.
|
||||
omitStages:
|
||||
- RequestReceived
|
||||
|
||||
rules:
|
||||
# Secrets/configmaps: record that access happened, never the contents.
|
||||
- level: Metadata
|
||||
resources:
|
||||
- group: ""
|
||||
resources: ["secrets", "configmaps"]
|
||||
|
||||
# Authn/authz decisions — the part an auditor actually asks about.
|
||||
- level: Metadata
|
||||
nonResourceURLs:
|
||||
- /apis*
|
||||
- /api*
|
||||
|
||||
# Mutations to workloads and policy: full request, so a diff is reconstructable.
|
||||
- level: Request
|
||||
verbs: ["create", "update", "patch", "delete"]
|
||||
resources:
|
||||
- group: ""
|
||||
resources: ["pods", "services", "serviceaccounts", "namespaces"]
|
||||
- group: "apps"
|
||||
- group: "networking.k8s.io"
|
||||
- group: "rbac.authorization.k8s.io"
|
||||
|
||||
# Everything else that changes state: metadata only.
|
||||
- level: Metadata
|
||||
verbs: ["create", "update", "patch", "delete"]
|
||||
|
||||
# Reads are dropped entirely — otherwise controller polling drowns the log.
|
||||
- level: None
|
||||
verbs: ["get", "list", "watch"]
|
||||
@@ -1,57 +0,0 @@
|
||||
# Cluster shape: one node, apiserver audit ON. Used by the `offline` profile.
|
||||
#
|
||||
# Audit is an apiserver flag, so it is fixed when the cluster is created —
|
||||
# changing it means `make cluster reset`, not a re-apply. That is why it is a
|
||||
# property of the cluster file rather than something switched at runtime.
|
||||
#
|
||||
# k8s >= 1.31 uses kubeadm v1beta4, where extraArgs is a LIST of name/value
|
||||
# pairs. The older map form is silently ignored — it does not error, audit
|
||||
# simply never turns on.
|
||||
#
|
||||
# Substituted by ctrl/cluster.sh: CLUSTER, NODE_IMAGE, HTTP_PORT, HOST_WORKDIR
|
||||
# (named without the ${...} braces so this line survives the substitution)
|
||||
kind: Cluster
|
||||
apiVersion: kind.x-k8s.io/v1alpha4
|
||||
name: ${CLUSTER}
|
||||
|
||||
containerdConfigPatches:
|
||||
- |-
|
||||
[plugins."io.containerd.grpc.v1.cri".registry]
|
||||
config_path = "/etc/containerd/certs.d"
|
||||
|
||||
kubeadmConfigPatches:
|
||||
- |
|
||||
kind: ClusterConfiguration
|
||||
apiServer:
|
||||
extraArgs:
|
||||
- name: audit-policy-file
|
||||
value: /etc/kubernetes/audit/policy.yaml
|
||||
- name: audit-log-path
|
||||
value: /var/log/kubernetes/audit.log
|
||||
- name: audit-log-maxage
|
||||
value: "7"
|
||||
extraVolumes:
|
||||
- name: audit-policy
|
||||
hostPath: /etc/kubernetes/audit
|
||||
mountPath: /etc/kubernetes/audit
|
||||
readOnly: true
|
||||
- name: audit-log
|
||||
hostPath: /var/log/kubernetes
|
||||
mountPath: /var/log/kubernetes
|
||||
readOnly: false
|
||||
|
||||
nodes:
|
||||
- role: control-plane
|
||||
image: ${NODE_IMAGE}
|
||||
# hostPath is resolved by the HOST dockerd, so this must be a host path even
|
||||
# when cluster.sh runs inside the installer container. HOST_WORKDIR says where
|
||||
# this rig lives on the host; bare on a host it is just the repo root.
|
||||
extraMounts:
|
||||
- hostPath: ${HOST_WORKDIR}/ctrl/k8s/audit-policy.yaml
|
||||
containerPath: /etc/kubernetes/audit/policy.yaml
|
||||
readOnly: true
|
||||
extraPortMappings:
|
||||
- containerPort: 30080
|
||||
hostPort: ${HTTP_PORT}
|
||||
listenAddress: "0.0.0.0"
|
||||
protocol: TCP
|
||||
@@ -1,55 +0,0 @@
|
||||
# Cluster shape: three nodes, apiserver audit ON. Used by the `client` profile —
|
||||
# the regulated-estate shape.
|
||||
#
|
||||
# Multi-node so taints, affinity and topology spread are real rather than
|
||||
# vacuously satisfied by a single node. It costs roughly 4-6 GB; run
|
||||
# `make cluster list` before starting this alongside other work.
|
||||
#
|
||||
# Substituted by ctrl/cluster.sh: CLUSTER, NODE_IMAGE, HTTP_PORT, HOST_WORKDIR
|
||||
# (named without the ${...} braces so this line survives the substitution)
|
||||
kind: Cluster
|
||||
apiVersion: kind.x-k8s.io/v1alpha4
|
||||
name: ${CLUSTER}
|
||||
|
||||
containerdConfigPatches:
|
||||
- |-
|
||||
[plugins."io.containerd.grpc.v1.cri".registry]
|
||||
config_path = "/etc/containerd/certs.d"
|
||||
|
||||
kubeadmConfigPatches:
|
||||
- |
|
||||
kind: ClusterConfiguration
|
||||
apiServer:
|
||||
extraArgs:
|
||||
- name: audit-policy-file
|
||||
value: /etc/kubernetes/audit/policy.yaml
|
||||
- name: audit-log-path
|
||||
value: /var/log/kubernetes/audit.log
|
||||
- name: audit-log-maxage
|
||||
value: "7"
|
||||
extraVolumes:
|
||||
- name: audit-policy
|
||||
hostPath: /etc/kubernetes/audit
|
||||
mountPath: /etc/kubernetes/audit
|
||||
readOnly: true
|
||||
- name: audit-log
|
||||
hostPath: /var/log/kubernetes
|
||||
mountPath: /var/log/kubernetes
|
||||
readOnly: false
|
||||
|
||||
nodes:
|
||||
- role: control-plane
|
||||
image: ${NODE_IMAGE}
|
||||
extraMounts:
|
||||
- hostPath: ${HOST_WORKDIR}/ctrl/k8s/audit-policy.yaml
|
||||
containerPath: /etc/kubernetes/audit/policy.yaml
|
||||
readOnly: true
|
||||
extraPortMappings:
|
||||
- containerPort: 30080
|
||||
hostPort: ${HTTP_PORT}
|
||||
listenAddress: "0.0.0.0"
|
||||
protocol: TCP
|
||||
- role: worker
|
||||
image: ${NODE_IMAGE}
|
||||
- role: worker
|
||||
image: ${NODE_IMAGE}
|
||||
@@ -1,23 +1,12 @@
|
||||
# Cluster shape: one node, no audit. Used by the `minimal` and `data` profiles.
|
||||
#
|
||||
# A TEMPLATE rather than a plain kind-config.yaml because a rig is copied and
|
||||
# renamed to make a second environment, and both the cluster name and the host
|
||||
# port follow the directory. A checked-in literal would make every copy collide
|
||||
# on both. ctrl/cluster.sh renders it with sed — not envsubst, which is
|
||||
# gettext-base and absent from a minimal Debian, and rig's whole premise is that
|
||||
# Docker is the only prerequisite.
|
||||
#
|
||||
# Substituted by ctrl/cluster.sh: CLUSTER, NODE_IMAGE, HTTP_PORT, HOST_WORKDIR
|
||||
# (named without the ${...} braces so this line survives the substitution)
|
||||
# Node count and audit are READ BACK from this file by lib/config.sh, so this
|
||||
# YAML is the source of truth for both — there is no second place to update.
|
||||
# The cluster. Add nodes or port mappings here, then `make cluster reset`.
|
||||
# ctrl/cluster.sh substitutes (sed): CLUSTER, NODE_IMAGE, HTTP_PORT, HOST_WORKDIR, OVERLAY_DIR
|
||||
# lib/config.sh reads the node count back from this file.
|
||||
# Notes: docs/notes/kind-config.md
|
||||
kind: Cluster
|
||||
apiVersion: kind.x-k8s.io/v1alpha4
|
||||
name: ${CLUSTER}
|
||||
|
||||
# Point containerd at a certs.d directory. registry.sh drops per-host hosts.toml
|
||||
# files in there afterwards, so switching registry mode never requires
|
||||
# recreating the cluster.
|
||||
# containerd reads per-host registry config from certs.d (written by registry.sh).
|
||||
containerdConfigPatches:
|
||||
- |-
|
||||
[plugins."io.containerd.grpc.v1.cri".registry]
|
||||
@@ -26,9 +15,7 @@ containerdConfigPatches:
|
||||
nodes:
|
||||
- role: control-plane
|
||||
image: ${NODE_IMAGE}
|
||||
# One NodePort bridged to the host; an in-cluster gateway owns it. There is
|
||||
# deliberately no ingress controller — they pin a narrow window of k8s
|
||||
# versions, and running a trailing-edge control plane is the point.
|
||||
# One NodePort bridged to the host, owned by an in-cluster gateway (no ingress controller).
|
||||
extraPortMappings:
|
||||
- containerPort: 30080
|
||||
hostPort: ${HTTP_PORT}
|
||||
|
||||
@@ -1,44 +1,40 @@
|
||||
# Shared config loading. Sourced, never executed.
|
||||
#
|
||||
# The ecosystem convention is that scripts are standalone with no shared log
|
||||
# library — that still holds. This file is not a logging lib; it is the single
|
||||
# definition of how the config layers compose, which every script has to agree
|
||||
# on exactly. Precedence, weakest first:
|
||||
#
|
||||
# ctrl/versions.env pinned toolchain + image digests (committed)
|
||||
# ctrl/env.d/<profile> cluster shape (committed)
|
||||
# ctrl/.env machine-local values and secrets (gitignored)
|
||||
# the caller's env `make cluster up PROFILE=client` (always wins)
|
||||
#
|
||||
# That last rule is why this is more than a few `source` lines: .env sets
|
||||
# PROFILE, so without snapshotting it would silently override the PROFILE the
|
||||
# user just typed on the command line.
|
||||
#
|
||||
# Shared config loading: how the config layers compose. Sourced, never executed.
|
||||
# Precedence, weakest first:
|
||||
# defaults < versions.env < env.d/<profile> < <overlay>/rig.env < .env < caller's env.
|
||||
# Run from ctrl/.
|
||||
# Notes: docs/notes/config.md
|
||||
|
||||
# Values a user can reasonably override per-invocation. Anything set in the
|
||||
# environment when load_config runs is restored after the files are read.
|
||||
# NODES and AUDIT are deliberately NOT here: they are properties of the chosen
|
||||
# ctrl/k8s/kind-config*.yaml.tpl and are read back out of it below, so there is
|
||||
# one place that decides the shape of the cluster rather than two that can drift.
|
||||
CONFIG_OVERRIDABLE="PROFILE CLUSTER K8S_VERSION KIND_CONFIG ADDONS
|
||||
# Per-invocation overrides: restored after the files are read, so the caller wins.
|
||||
# NODES is deliberately not here (read from the kind config).
|
||||
CONFIG_OVERRIDABLE="PROFILE OVERLAY CLUSTER K8S_VERSION KIND_CONFIG ADDONS
|
||||
REGISTRY_MODE INGRESS_MODE DNS_MODE TILT_PORT
|
||||
SOURCE ARCH DEPS_SOURCE HTTP_PORT HTTPS_PORT"
|
||||
SOURCE ARCH DEPS_SOURCE HTTP_PORT HTTPS_PORT
|
||||
REGISTRY_PORT MANIFESTS_DIR"
|
||||
|
||||
# The containing folder's name, reduced to something kind accepts as a cluster
|
||||
# name (a DNS label: lowercase alphanumerics and dashes). Run from ctrl/, so the
|
||||
# repo root is the parent.
|
||||
# rig's own example, used when no overlay is named (relative to rig's root).
|
||||
DEFAULT_OVERLAY=examples/starter
|
||||
|
||||
# A rig-root-relative path as seen from ctrl/; absolute paths pass through.
|
||||
_from_ctrl() { case "$1" in /*) echo "$1" ;; *) echo "../$1" ;; esac; }
|
||||
|
||||
# The same, absolute. Empty if it does not exist.
|
||||
_abs_from_ctrl() { (cd "$(_from_ctrl "$1")" 2>/dev/null && pwd); }
|
||||
|
||||
# The environment's folder — the overlay's when one is named, else rig's —
|
||||
# reduced to a DNS label kind accepts as a cluster name.
|
||||
default_cluster_name() {
|
||||
local n
|
||||
if [ -n "${OVERLAY:-}" ]; then
|
||||
n=$(basename "$(_abs_from_ctrl "$OVERLAY_DIR")")
|
||||
else
|
||||
n=$(basename "$(cd .. && pwd)")
|
||||
fi
|
||||
n=$(echo "$n" | tr '[:upper:]' '[:lower:]' | tr -c 'a-z0-9-' '-')
|
||||
n=$(echo "$n" | sed 's/^-*//; s/-*$//')
|
||||
echo "${n:-rig}"
|
||||
}
|
||||
|
||||
# Base of this environment's 10-port block. cksum is used rather than $RANDOM or
|
||||
# bash hashing because it is POSIX and returns the same value on every machine,
|
||||
# which is what makes the block reproducible instead of merely unique.
|
||||
# Base of this environment's 10-port block; cksum so it is the same on every machine.
|
||||
derive_port_base() {
|
||||
local h; h=$(printf '%s' "$1" | cksum | awk '{print $1}')
|
||||
echo $((20000 + (h % 200) * 10))
|
||||
@@ -56,43 +52,119 @@ load_config() {
|
||||
|
||||
set -a
|
||||
source ./versions.env
|
||||
[ -f ./.env ] && source ./.env
|
||||
# RIG_PORTABLE skips the machine-local .env (set by config_snapshot for kits).
|
||||
if [ -z "${RIG_PORTABLE:-}" ] && [ -f ./.env ]; then source ./.env; fi
|
||||
set +a
|
||||
|
||||
# Re-apply overrides now so PROFILE is the caller's before we pick the file.
|
||||
_config_restore "$saved"
|
||||
|
||||
local profile="${PROFILE:-minimal}"
|
||||
# A profile is optional; naming one that does not exist is an error.
|
||||
local profile="${PROFILE:-}" layered=""
|
||||
if [ -n "$profile" ] && [ "$profile" != default ]; then
|
||||
if [ ! -f "./env.d/${profile}.env" ]; then
|
||||
echo "no such profile: env.d/${profile}.env" >&2
|
||||
echo "available: $(ls env.d/*.env 2>/dev/null | xargs -n1 basename | sed 's/\.env$//' | tr '\n' ' ')" >&2
|
||||
if [ -d "../examples/${profile}" ]; then
|
||||
echo " it is an example overlay now: OVERLAY=examples/${profile}" >&2
|
||||
fi
|
||||
echo "available: $(config_profiles | tr '\n' ' ')" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
set -a
|
||||
source "./env.d/${profile}.env"
|
||||
[ -f ./.env ] && source ./.env
|
||||
set +a
|
||||
layered=1
|
||||
fi
|
||||
|
||||
# The overlay: one folder, outside rig, holding a use case (docs/notes/overlay.md).
|
||||
# Named ones must exist; with none named, rig's own example is used if present.
|
||||
OVERLAY_DIR=""
|
||||
if [ -n "${OVERLAY:-}" ]; then
|
||||
OVERLAY_DIR="${OVERLAY%/}"
|
||||
if [ ! -d "$(_from_ctrl "$OVERLAY_DIR")" ]; then
|
||||
echo "no overlay at OVERLAY=${OVERLAY} (relative to rig's folder, or absolute)" >&2
|
||||
exit 1
|
||||
fi
|
||||
elif [ -d "../${DEFAULT_OVERLAY}" ]; then
|
||||
OVERLAY_DIR="$DEFAULT_OVERLAY"
|
||||
fi
|
||||
# Its rig.env may not choose the profile or the overlay (both are chosen before
|
||||
# it loads), and the paths it sets are relative to the overlay.
|
||||
local ov_env="" m_before k_before
|
||||
if [ -n "$OVERLAY_DIR" ]; then ov_env="$(_from_ctrl "$OVERLAY_DIR")/rig.env"; fi
|
||||
if [ -n "$ov_env" ] && [ -f "$ov_env" ]; then
|
||||
if grep -qE '^[[:space:]]*(export[[:space:]]+)?(PROFILE|OVERLAY)=' "$ov_env"; then
|
||||
echo "$ov_env: an overlay's rig.env cannot set PROFILE or OVERLAY (they choose it)" >&2
|
||||
exit 1
|
||||
fi
|
||||
m_before="${MANIFESTS_DIR-}" k_before="${KIND_CONFIG-}"
|
||||
set -a
|
||||
source "$ov_env"
|
||||
set +a
|
||||
if [ "${MANIFESTS_DIR-}" != "$m_before" ]; then
|
||||
case "$MANIFESTS_DIR" in /*|none|"") ;; *) MANIFESTS_DIR="${OVERLAY_DIR}/${MANIFESTS_DIR}" ;; esac
|
||||
fi
|
||||
if [ "${KIND_CONFIG-}" != "$k_before" ]; then
|
||||
case "$KIND_CONFIG" in /*|"") ;; *) KIND_CONFIG="$(dirname "$ov_env")/${KIND_CONFIG}" ;; esac
|
||||
fi
|
||||
layered=1
|
||||
fi
|
||||
|
||||
# The machine and the caller still win over both.
|
||||
if [ -n "$layered" ]; then
|
||||
set -a
|
||||
if [ -z "${RIG_PORTABLE:-}" ] && [ -f ./.env ]; then source ./.env; fi
|
||||
set +a
|
||||
_config_restore "$saved"
|
||||
fi
|
||||
|
||||
# Identity follows the FOLDER, so copying this directory somewhere else and
|
||||
# renaming it yields a distinct environment with no further edits. Without
|
||||
# this, two copies would share one cluster and `make cluster down` in either
|
||||
# would destroy the other's.
|
||||
# The defaults a profile would otherwise have to supply. Weakest of all: a
|
||||
# profile, ctrl/.env and the caller each override them.
|
||||
PROFILE_NAME="${PROFILE_NAME:-default}"
|
||||
ADDONS="${ADDONS-}"
|
||||
# local, not none: with no registry an unqualified image name means
|
||||
# docker.io/library/<name>, and a default must not make that disclosure.
|
||||
REGISTRY_MODE="${REGISTRY_MODE:-local}"
|
||||
INGRESS_MODE="${INGRESS_MODE:-hostport}"
|
||||
DNS_MODE="${DNS_MODE:-hosts}"
|
||||
# The newest node image versions.env pins, found rather than restated, so
|
||||
# bumping the pins moves the default with them.
|
||||
if [ -z "${K8S_VERSION:-}" ]; then
|
||||
K8S_VERSION=$(compgen -v NODE_IMAGE_v | sort -V | tail -1)
|
||||
K8S_VERSION="${K8S_VERSION#NODE_IMAGE_}"
|
||||
fi
|
||||
|
||||
# Identity follows the folder, so a renamed copy is a distinct environment.
|
||||
CLUSTER="${CLUSTER:-$(default_cluster_name)}"
|
||||
KUBECONTEXT="kind-${CLUSTER}"
|
||||
|
||||
# Host ports are a single shared namespace, so unlike the cluster name they
|
||||
# cannot just follow the directory — they have to be spread out. Anything
|
||||
# already set (ctrl/.env, a profile, the command line) wins; only the gaps
|
||||
# are filled. See ports.sh for the reasoning.
|
||||
# Host ports: fill only the gaps from the derived block; anything already set wins.
|
||||
local base; base=$(derive_port_base "$CLUSTER")
|
||||
HTTP_PORT="${HTTP_PORT:-$base}"
|
||||
HTTPS_PORT="${HTTPS_PORT:-$((base + 1))}"
|
||||
TILT_PORT="${TILT_PORT:-$((base + 2))}"
|
||||
REGISTRY_PORT="${REGISTRY_PORT:-$((base + 3))}"
|
||||
|
||||
# Where the workload's manifests live, relative to rig's folder (or absolute):
|
||||
# the overlay's k8s/overlays/dev unless something names another. `none`: rig
|
||||
# applies none (the overlay's Tiltfile does). A named folder must exist.
|
||||
if [ "${MANIFESTS_DIR:-}" = ctrl/k8s/overlays/dev ] && [ ! -d ../ctrl/k8s/overlays/dev ]; then
|
||||
# The old default, pinned by an older .env.example; rig's examples moved.
|
||||
STALE_MANIFESTS_DIR="$MANIFESTS_DIR"
|
||||
MANIFESTS_DIR=""
|
||||
fi
|
||||
if [ -z "${MANIFESTS_DIR:-}" ] && [ -n "$OVERLAY_DIR" ] \
|
||||
&& [ -d "$(_from_ctrl "$OVERLAY_DIR")/k8s/overlays/dev" ]; then
|
||||
MANIFESTS_DIR="$OVERLAY_DIR/k8s/overlays/dev"
|
||||
fi
|
||||
MANIFESTS_DIR="${MANIFESTS_DIR:-}"
|
||||
if [ "$MANIFESTS_DIR" = none ]; then
|
||||
MANIFESTS_DIR=""
|
||||
elif [ -n "$MANIFESTS_DIR" ] && [ ! -d "$(_from_ctrl "$MANIFESTS_DIR")" ]; then
|
||||
echo "no manifests at MANIFESTS_DIR=${MANIFESTS_DIR} (relative to rig's folder, or absolute)" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Profiles name a k8s minor (v1_36); versions.env holds the pinned digest.
|
||||
local var="NODE_IMAGE_${K8S_VERSION}"
|
||||
NODE_IMAGE="${!var:-}"
|
||||
@@ -101,48 +173,39 @@ load_config() {
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# The cluster's shape is a file in ctrl/k8s/, named by the profile. Adding a
|
||||
# shape is adding a file; there is no dispatcher to edit.
|
||||
#
|
||||
# A host that needs its own shape — extra port mappings, more nodes — passes
|
||||
# an absolute path instead, and rig renders it exactly like one of its own:
|
||||
# ${CLUSTER} and ${NODE_IMAGE} are substituted either way. The shape stays in
|
||||
# the host's tree, because what a host's cluster needs is the host's business;
|
||||
# rig only knows how to build whatever it is handed.
|
||||
KIND_CONFIG="${KIND_CONFIG:-kind-config.yaml.tpl}"
|
||||
case "$KIND_CONFIG" in
|
||||
/*) KIND_CONFIG_PATH="$KIND_CONFIG"; KIND_CONFIG_SHOWN="$KIND_CONFIG" ;;
|
||||
*) KIND_CONFIG_PATH="./k8s/${KIND_CONFIG}"; KIND_CONFIG_SHOWN="ctrl/k8s/${KIND_CONFIG}" ;;
|
||||
esac
|
||||
if [ ! -f "$KIND_CONFIG_PATH" ]; then
|
||||
echo "no such cluster shape: ${KIND_CONFIG_SHOWN}" >&2
|
||||
echo "rig's own: $(ls k8s/kind-config*.yaml.tpl 2>/dev/null | xargs -n1 basename | tr '\n' ' ')" >&2
|
||||
echo "or pass an absolute path to a shape of your own" >&2
|
||||
# The cluster is one file: the overlay's kind-config.yaml.tpl if it has one,
|
||||
# else rig's k8s/kind-config.yaml.tpl. KIND_CONFIG is "use this file instead",
|
||||
# for a project that builds its own cluster through rig (relative to ctrl/, or absolute).
|
||||
if [ -z "${KIND_CONFIG:-}" ] && [ -n "$OVERLAY_DIR" ] \
|
||||
&& [ -f "$(_from_ctrl "$OVERLAY_DIR")/kind-config.yaml.tpl" ]; then
|
||||
KIND_CONFIG="$(_from_ctrl "$OVERLAY_DIR")/kind-config.yaml.tpl"
|
||||
fi
|
||||
KIND_CONFIG="${KIND_CONFIG:-./k8s/kind-config.yaml.tpl}"
|
||||
if [ ! -f "$KIND_CONFIG" ]; then
|
||||
echo "no kind config at KIND_CONFIG=${KIND_CONFIG}" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Read the shape back out of the YAML rather than trusting a profile to
|
||||
# restate it. check.sh sizes the memory warning on NODES, and cluster.sh
|
||||
# prints AUDIT before spending minutes building something that cannot be
|
||||
# changed afterwards — both would mislead if the numbers drifted.
|
||||
NODES=$(grep -c '^ - role:' "$KIND_CONFIG_PATH")
|
||||
if grep -q 'audit-policy-file' "$KIND_CONFIG_PATH"; then AUDIT=on; else AUDIT=off; fi
|
||||
# Read the node count back out of the file rather than restating it:
|
||||
# check.sh and the memory tool size their budget on NODES.
|
||||
NODES=$(grep -c '^ - role:' "$KIND_CONFIG")
|
||||
|
||||
# Measured MB per node (cluster alone, errs high for workers); shared by
|
||||
# check.sh, the memory tool and standalone kits.
|
||||
NODE_MB=800
|
||||
}
|
||||
|
||||
# Render a cluster shape to stdout. sed rather than envsubst: envsubst is
|
||||
# gettext-base, absent from a minimal Debian, and Docker is meant to be the only
|
||||
# prerequisite. The variable list is explicit so a template cannot quietly start
|
||||
# depending on something the caller does not set.
|
||||
#
|
||||
# hostPath entries are resolved by the HOST dockerd, so HOST_WORKDIR must stay a
|
||||
# host path even when this runs inside the installer container.
|
||||
# Render the kind config to stdout with sed (not envsubst) over an explicit variable list.
|
||||
# HOST_WORKDIR and OVERLAY_DIR must be host paths: the host dockerd resolves hostPath entries.
|
||||
render_kind_config() {
|
||||
local host_workdir="${HOST_WORKDIR:-$(cd .. && pwd)}"
|
||||
local host_workdir="${HOST_WORKDIR:-$(cd .. && pwd)}" overlay_dir=""
|
||||
if [ -n "$OVERLAY_DIR" ]; then overlay_dir=$(_abs_from_ctrl "$OVERLAY_DIR"); fi
|
||||
sed -e "s|\${CLUSTER}|${CLUSTER}|g" \
|
||||
-e "s|\${NODE_IMAGE}|${NODE_IMAGE}|g" \
|
||||
-e "s|\${HTTP_PORT}|${HTTP_PORT}|g" \
|
||||
-e "s|\${HOST_WORKDIR}|${host_workdir}|g" \
|
||||
"$KIND_CONFIG_PATH"
|
||||
-e "s|\${OVERLAY_DIR}|${overlay_dir}|g" \
|
||||
"$KIND_CONFIG"
|
||||
}
|
||||
|
||||
_config_restore() {
|
||||
@@ -156,3 +219,87 @@ _config_restore() {
|
||||
# line would otherwise make this return 1 and trip `set -e` in the caller.
|
||||
return 0
|
||||
}
|
||||
|
||||
# ── what a standalone kit needs to know ────────────────────────────────────
|
||||
# The questions ctrl/standalone.sh asks, so it never knows how config is stored.
|
||||
|
||||
# Every configuration rig can run as, one per line: each profile, or `default`
|
||||
# when there are none. Never empty.
|
||||
config_profiles() {
|
||||
local f found=""
|
||||
for f in ./env.d/*.env; do
|
||||
[ -e "$f" ] || continue
|
||||
f=${f##*/}; echo "${f%.env}"; found=1
|
||||
done
|
||||
[ -n "$found" ] || echo default
|
||||
}
|
||||
|
||||
# What load_config sets, minus the machine-local layer, as `declare -p` lines.
|
||||
# Usage: config_snapshot <profile> | --current (found by difference, not a list)
|
||||
config_snapshot() {
|
||||
local _rig_snap_choices
|
||||
if [ "$1" = --current ]; then
|
||||
_rig_snap_choices=$( (
|
||||
load_config >/dev/null || exit 1
|
||||
for _rig_snap_n in $CONFIG_OVERRIDABLE; do
|
||||
if [ -n "${!_rig_snap_n+x}" ]; then printf 'export %s=%q\n' "$_rig_snap_n" "${!_rig_snap_n}"; fi
|
||||
done
|
||||
) ) || return 1
|
||||
else
|
||||
_rig_snap_choices="export PROFILE=$(printf '%q' "$1")"
|
||||
fi
|
||||
(
|
||||
# Nothing from the caller's shell may leak into a kit.
|
||||
for _rig_snap_n in $CONFIG_OVERRIDABLE; do unset "$_rig_snap_n"; done
|
||||
declare -A _rig_snap_was=()
|
||||
for _rig_snap_n in $(compgen -v); do
|
||||
_rig_snap_was[$_rig_snap_n]="${!_rig_snap_n-}"
|
||||
done
|
||||
eval "$_rig_snap_choices"
|
||||
RIG_PORTABLE=1 load_config >/dev/null
|
||||
for _rig_snap_n in $(compgen -v); do
|
||||
case "$_rig_snap_n" in
|
||||
_rig_snap_*|RIG_PORTABLE|BASH*|FUNCNAME|PIPESTATUS|LINENO|RANDOM|SRANDOM|\
|
||||
SECONDS|EPOCH*|HISTCMD|COLUMNS|LINES|PWD|OLDPWD|_|SHLVL|OPTIND|OPTERR) continue ;;
|
||||
esac
|
||||
if [ -z "${_rig_snap_was[$_rig_snap_n]+x}" ] \
|
||||
|| [ "${_rig_snap_was[$_rig_snap_n]}" != "${!_rig_snap_n-}" ]; then
|
||||
declare -p "$_rig_snap_n"
|
||||
fi
|
||||
done
|
||||
)
|
||||
}
|
||||
|
||||
# The profile this machine runs, as load_config resolves it here.
|
||||
config_current_profile() { ( load_config >/dev/null && echo "$PROFILE_NAME" ); }
|
||||
|
||||
# Names (never values) of .env keys an export does not carry, e.g. credentials.
|
||||
config_left_out() {
|
||||
[ -f ./.env ] || return 0
|
||||
local k
|
||||
for k in $(sed -nE 's/^[[:space:]]*(export[[:space:]]+)?([A-Za-z_][A-Za-z0-9_]*)=.*/\2/p' ./.env | sort -u); do
|
||||
case " $(echo $CONFIG_OVERRIDABLE) " in
|
||||
*" $k "*) ;;
|
||||
*) echo "$k" ;;
|
||||
esac
|
||||
done
|
||||
}
|
||||
|
||||
# Print a load_config with a resolution frozen in, for a standalone kit to carry.
|
||||
# The caller's env still wins; derived values (e.g. ports) stay fixed.
|
||||
config_freeze() {
|
||||
local snap
|
||||
snap=$(config_snapshot "$1") || return 1
|
||||
cat <<'EOF'
|
||||
load_config() {
|
||||
local k saved=""
|
||||
for k in $CONFIG_OVERRIDABLE; do
|
||||
if [ -n "${!k+x}" ]; then saved+="$k=$(printf '%q' "${!k}")"$'\n'; fi
|
||||
done
|
||||
EOF
|
||||
printf '%s\n' "$snap" | sed -E 's/^declare --* / declare -g /; s/^declare -([a-zA-Z]+) / declare -g\1 /'
|
||||
cat <<'EOF'
|
||||
_config_restore "$saved"
|
||||
}
|
||||
EOF
|
||||
}
|
||||
|
||||
671
rig/ctrl/mem.sh
671
rig/ctrl/mem.sh
@@ -1,28 +1,24 @@
|
||||
#!/usr/bin/env bash
|
||||
# What memory this machine has, what is left, and — where there is one — what
|
||||
# cap is holding it there.
|
||||
#
|
||||
# Runs on native Linux and under WSL, because rig is developed on one and used
|
||||
# on the other. The difference is not cosmetic: on WSL the memory you see is a
|
||||
# VM allocation that can be raised, and the commonest failure is raising it
|
||||
# without restarting, so the number on disk and the number in /proc disagree.
|
||||
# On native Linux there is no such cap and pretending otherwise sends you to a
|
||||
# file that does not exist.
|
||||
#
|
||||
# This reports and instructs. It never writes a .wslconfig — applying one costs
|
||||
# a full VM restart that takes every shell, mount and container with it, and
|
||||
# choosing that moment is yours.
|
||||
#
|
||||
# `backup` exists so `restore` has something to read: back up, hand-edit
|
||||
# following the printed instruction, restore if it goes wrong. Both are
|
||||
# WSL-only, because .wslconfig is the only thing here worth backing up.
|
||||
#
|
||||
# Usage: mem.sh status | backup | restore
|
||||
# rig:standalone rigmini status
|
||||
# rig's memory tool (also generated as rigmini.sh): what the machine advertises vs. what it survives.
|
||||
# Usage: mem.sh status | push [--to GB] [--to-oom] | all [--budget GB] | backup | restore (WSL)
|
||||
# Notes: docs/notes/mem.md
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")"
|
||||
source ./lib/config.sh
|
||||
|
||||
# Windows outside WSL — Git Bash, MSYS, Cygwin — looks close enough to work and
|
||||
# then fails in a pile of confusing ways: no /proc, no docker socket, none of
|
||||
# the tooling. Detectable, so name it instead.
|
||||
# ── defaults ───────────────────────────────────────────────────────────────
|
||||
|
||||
STEP_MB=0 # per allocation; 0 means scale it to the ceiling. See push().
|
||||
STEP_EXPLICIT=no # whether --step was given, which turns the scaling off.
|
||||
TO_MB="" # --to: stop here regardless. Empty means no hard cap.
|
||||
TO_OOM=no # --to-oom: opt in to running until the kernel intervenes.
|
||||
BUDGET_GB="" # --budget; empty means what this profile's cluster needs, from rig.
|
||||
BUDGET_EXPLICIT=no # whether --budget was given, which retires the guess below.
|
||||
|
||||
# ── platform ───────────────────────────────────────────────────────────────
|
||||
|
||||
# Refuse Git Bash / MSYS / Cygwin and kernels without /proc, with a clear message.
|
||||
require_linux() {
|
||||
case "$(uname -s)" in
|
||||
MINGW*|MSYS*|CYGWIN*)
|
||||
@@ -36,29 +32,141 @@ If WSL is not installed yet, from an elevated PowerShell or Command Prompt:
|
||||
That enables Windows features and needs a reboot, so it is not something this
|
||||
script will do for you. Afterwards, open the Linux shell it installs and run
|
||||
this from there.
|
||||
|
||||
See "Starting from plain Windows" in README.md.
|
||||
EOF
|
||||
exit 1 ;;
|
||||
esac
|
||||
|
||||
# Everything below reads /proc. Without it there is nothing to measure, and
|
||||
# failing here beats printing a page of empty fields.
|
||||
if [ ! -r /proc/meminfo ]; then
|
||||
echo "no readable /proc/meminfo — this needs a Linux kernel." >&2
|
||||
echo "On macOS or a BSD none of the numbers below exist." >&2
|
||||
exit 1
|
||||
fi
|
||||
}
|
||||
|
||||
is_wsl() { grep -qi microsoft /proc/version 2>/dev/null; }
|
||||
|
||||
require_wsl() {
|
||||
if ! is_wsl; then
|
||||
echo "$1 acts on .wslconfig, which only exists under WSL." >&2
|
||||
echo "This is native Linux — there is no VM allocation to save or roll back." >&2
|
||||
echo "Use 'mem.sh status' to see what the machine actually has." >&2
|
||||
exit 1
|
||||
is_container() {
|
||||
[ -f /.dockerenv ] && return 0
|
||||
grep -qE '(docker|containerd|kubepods|lxc|podman)' /proc/1/cgroup 2>/dev/null
|
||||
}
|
||||
|
||||
platform() {
|
||||
if is_wsl; then echo WSL
|
||||
elif is_container; then echo container
|
||||
else echo "native linux"
|
||||
fi
|
||||
}
|
||||
|
||||
# ── reading memory ─────────────────────────────────────────────────────────
|
||||
|
||||
mb() { echo $(( $(awk "/^$1:/{print \$2}" /proc/meminfo) / 1024 )); }
|
||||
|
||||
# /mnt/c/Users can hold several real accounts — a renamed login leaves the old
|
||||
# directory behind — so picking the first alphabetically is a coin toss. Ask
|
||||
# Windows, then fall back to whichever profile actually owns a config.
|
||||
# MemAvailable arrived in kernel 3.14. Older kernels — and they turn up on
|
||||
# corporate images — need the estimate it replaced, which is worse but not wrong.
|
||||
avail_meminfo_mb() {
|
||||
if grep -q '^MemAvailable:' /proc/meminfo; then
|
||||
mb MemAvailable
|
||||
else
|
||||
awk '/^(MemFree|Buffers|Cached):/{t+=$2} END{print int(t/1024)}' /proc/meminfo
|
||||
fi
|
||||
}
|
||||
|
||||
# This cgroup's limit/usage files, set once by find_cgroup (cheap for the poll loop).
|
||||
CG_MAX_FILE=""
|
||||
CG_CUR_FILE=""
|
||||
CG_VERSION=""
|
||||
|
||||
find_cgroup() {
|
||||
local rel
|
||||
|
||||
# Top of tree first (right inside a container), then this shell's own slice
|
||||
# from /proc/self/cgroup (right on a host).
|
||||
if [ -r /sys/fs/cgroup/memory.max ]; then
|
||||
CG_VERSION=v2
|
||||
CG_MAX_FILE=/sys/fs/cgroup/memory.max
|
||||
CG_CUR_FILE=/sys/fs/cgroup/memory.current
|
||||
elif [ -r /sys/fs/cgroup/memory/memory.limit_in_bytes ]; then
|
||||
CG_VERSION=v1
|
||||
CG_MAX_FILE=/sys/fs/cgroup/memory/memory.limit_in_bytes
|
||||
CG_CUR_FILE=/sys/fs/cgroup/memory/memory.usage_in_bytes
|
||||
fi
|
||||
|
||||
rel=$(awk -F: '$1=="0"{print $3; exit}' /proc/self/cgroup 2>/dev/null || true)
|
||||
if [ -n "$rel" ] && [ "$rel" != "/" ] && [ -r "/sys/fs/cgroup${rel}/memory.max" ]; then
|
||||
CG_VERSION=v2
|
||||
CG_MAX_FILE="/sys/fs/cgroup${rel}/memory.max"
|
||||
CG_CUR_FILE="/sys/fs/cgroup${rel}/memory.current"
|
||||
return 0
|
||||
fi
|
||||
|
||||
rel=$(awk -F: '$2 ~ /(^|,)memory(,|$)/{print $3; exit}' /proc/self/cgroup 2>/dev/null || true)
|
||||
if [ -n "$rel" ] && [ "$rel" != "/" ] \
|
||||
&& [ -r "/sys/fs/cgroup/memory${rel}/memory.limit_in_bytes" ]; then
|
||||
CG_VERSION=v1
|
||||
CG_MAX_FILE="/sys/fs/cgroup/memory${rel}/memory.limit_in_bytes"
|
||||
CG_CUR_FILE="/sys/fs/cgroup/memory${rel}/memory.usage_in_bytes"
|
||||
fi
|
||||
return 0
|
||||
}
|
||||
|
||||
# The cap in MB, or "" when unlimited ("max", or any value >= MemTotal).
|
||||
cgroup_cap_mb() {
|
||||
local raw cap
|
||||
[ -n "$CG_MAX_FILE" ] && [ -r "$CG_MAX_FILE" ] || { echo ""; return 0; }
|
||||
raw=$(cat "$CG_MAX_FILE" 2>/dev/null || echo max)
|
||||
[ "$raw" = "max" ] && { echo ""; return 0; }
|
||||
case "$raw" in ''|*[!0-9]*) echo ""; return 0 ;; esac
|
||||
cap=$((raw / 1024 / 1024))
|
||||
[ "$cap" -ge "$(mb MemTotal)" ] && { echo ""; return 0; }
|
||||
echo "$cap"
|
||||
}
|
||||
|
||||
cgroup_used_mb() {
|
||||
local raw
|
||||
[ -n "$CG_CUR_FILE" ] && [ -r "$CG_CUR_FILE" ] || { echo ""; return 0; }
|
||||
raw=$(cat "$CG_CUR_FILE" 2>/dev/null || echo "")
|
||||
case "$raw" in ''|*[!0-9]*) echo ""; return 0 ;; esac
|
||||
echo $((raw / 1024 / 1024))
|
||||
}
|
||||
|
||||
# ulimit -v is a per-process address-space cap. It stops YOU long before the box
|
||||
# does, and because it is inherited from a login shell it is easy to hit without
|
||||
# knowing it is set.
|
||||
ulimit_v_mb() {
|
||||
local v; v=$(ulimit -v 2>/dev/null || echo unlimited)
|
||||
[ "$v" = "unlimited" ] && { echo ""; return 0; }
|
||||
case "$v" in ''|*[!0-9]*) echo ""; return 0 ;; esac
|
||||
echo $((v / 1024))
|
||||
}
|
||||
|
||||
# The number everything else is about: the lowest of the things that can stop
|
||||
# you. Printed at the end of `status` and used as the sanity bound in `push`.
|
||||
effective_ceiling_mb() {
|
||||
local c; c=$(mb MemTotal)
|
||||
local cap; cap=$(cgroup_cap_mb)
|
||||
local ul; ul=$(ulimit_v_mb)
|
||||
[ -n "$cap" ] && [ "$cap" -lt "$c" ] && c="$cap"
|
||||
[ -n "$ul" ] && [ "$ul" -lt "$c" ] && c="$ul"
|
||||
echo "$c"
|
||||
}
|
||||
|
||||
# Room left right now: cgroup cap minus usage when capped, else MemAvailable.
|
||||
headroom_mb() {
|
||||
local cap used
|
||||
cap=$(cgroup_cap_mb)
|
||||
used=$(cgroup_used_mb)
|
||||
if [ -n "$cap" ] && [ -n "$used" ]; then
|
||||
echo $(( cap - used ))
|
||||
else
|
||||
avail_meminfo_mb
|
||||
fi
|
||||
}
|
||||
|
||||
# ── status ─────────────────────────────────────────────────────────────────
|
||||
|
||||
# Ask Windows for %USERPROFILE%; fall back to whichever profile owns a .wslconfig.
|
||||
wslconfig_path() {
|
||||
local profile winpath found
|
||||
profile=$(cmd.exe /c "echo %USERPROFILE%" 2>/dev/null | tr -d "\r\n" || true)
|
||||
@@ -66,113 +174,220 @@ wslconfig_path() {
|
||||
""|*%*) ;;
|
||||
*) winpath=$(wslpath -u "$profile" 2>/dev/null || true)
|
||||
if [ -n "$winpath" ] && [ -d "$winpath" ]; then
|
||||
echo "$winpath/.wslconfig"; return
|
||||
echo "$winpath/.wslconfig"; return 0
|
||||
fi ;;
|
||||
esac
|
||||
|
||||
found=$(ls -d /mnt/c/Users/*/.wslconfig 2>/dev/null | head -1 || true)
|
||||
if [ -n "$found" ]; then echo "$found"; return; fi
|
||||
|
||||
echo "cannot tell which Windows profile owns .wslconfig. Candidates:" >&2
|
||||
ls -d /mnt/c/Users/*/ 2>/dev/null \
|
||||
| grep -viE "/(All Users|Default|Default User|Public)/$" | sed "s/^/ /" >&2
|
||||
exit 1
|
||||
}
|
||||
|
||||
configured_memory() {
|
||||
[ -r "$1" ] || { echo ""; return; }
|
||||
sed -n 's/^[[:space:]]*memory[[:space:]]*=[[:space:]]*//p' "$1" | tail -1 | tr -d '[:space:]'
|
||||
}
|
||||
|
||||
# "9GB" / "8192MB" / "9G" -> MB, so it can be compared with /proc/meminfo.
|
||||
to_mb() {
|
||||
local v="${1^^}" n
|
||||
n=$(echo "$v" | tr -dc '0-9')
|
||||
[ -n "$n" ] || { echo ""; return; }
|
||||
case "$v" in
|
||||
*GB|*G) echo $(( n * 1024 )) ;;
|
||||
*MB|*M) echo "$n" ;;
|
||||
*) echo $(( n / 1024 / 1024 )) ;;
|
||||
esac
|
||||
[ -n "$found" ] && echo "$found"
|
||||
return 0
|
||||
}
|
||||
|
||||
hogs() {
|
||||
echo " holding the most:"
|
||||
ps -eo rss,comm --sort=-rss 2>/dev/null | awk 'NR>1 && NR<=6 {printf " %6.0f MB %s\n", $1/1024, $2}'
|
||||
ps -eo rss,comm --sort=-rss 2>/dev/null \
|
||||
| awk 'NR>1 && NR<=6 {printf " %6.0f MB %s\n", $1/1024, $2}'
|
||||
return 0
|
||||
}
|
||||
|
||||
status() {
|
||||
local total avail swap_total swap_free
|
||||
total=$(mb MemTotal); avail=$(mb MemAvailable)
|
||||
swap_total=$(mb SwapTotal); swap_free=$(mb SwapFree)
|
||||
local total avail swap_total swap_free cap ul cur
|
||||
|
||||
if is_wsl; then
|
||||
local cfg conf conf_mb
|
||||
cfg=$(wslconfig_path)
|
||||
conf=$(configured_memory "$cfg")
|
||||
echo "platform WSL"
|
||||
echo "config $cfg"
|
||||
if [ -n "$conf" ]; then
|
||||
conf_mb=$(to_mb "$conf")
|
||||
echo "configured $conf (${conf_mb} MB)"
|
||||
echo "host"
|
||||
echo " platform $(platform)"
|
||||
echo " kernel $(uname -r)"
|
||||
[ -r /etc/os-release ] && \
|
||||
echo " distro $(sed -n 's/^PRETTY_NAME="\(.*\)"/\1/p' /etc/os-release)"
|
||||
echo " cpu $(getconf _NPROCESSORS_ONLN 2>/dev/null || echo '?') online, load $(cut -d' ' -f1-3 /proc/loadavg)"
|
||||
|
||||
# ── the caps first, because they decide what the totals below are worth ──
|
||||
echo
|
||||
echo "caps"
|
||||
cap=$(cgroup_cap_mb)
|
||||
if [ -n "$cap" ]; then
|
||||
cur=$(cgroup_used_mb)
|
||||
echo " cgroup ${cap} MB (${CG_VERSION}, ${CG_CUR_FILE##*/} says ${cur:-?} MB used)"
|
||||
echo " ! /proc/meminfo below describes the HOST, not this cgroup."
|
||||
echo " $(mb MemTotal) MB total is not yours; ${cap} MB is."
|
||||
elif [ -n "$CG_VERSION" ]; then
|
||||
echo " cgroup none (${CG_VERSION} present, no memory limit set)"
|
||||
else
|
||||
echo " cgroup no memory controller found"
|
||||
fi
|
||||
|
||||
ul=$(ulimit_v_mb)
|
||||
if [ -n "$ul" ]; then
|
||||
echo " ! ulimit -v ${ul} MB — a per-process cap, inherited from your shell"
|
||||
echo " it stops this process long before the machine runs out"
|
||||
else
|
||||
conf_mb=""
|
||||
echo "configured (no memory= set — WSL defaults to 50% of host RAM, or 8GB, whichever is less)"
|
||||
echo " ulimit -v unlimited"
|
||||
fi
|
||||
echo "booted ${total} MB"
|
||||
|
||||
# Overcommit mode decides whether limits show as failed mallocs or OOM kills.
|
||||
local om or_
|
||||
om=$(cat /proc/sys/vm/overcommit_memory 2>/dev/null || echo '?')
|
||||
or_=$(cat /proc/sys/vm/overcommit_ratio 2>/dev/null || echo '?')
|
||||
case "$om" in
|
||||
0) echo " overcommit 0 heuristic — allocations are granted on a guess," ;;
|
||||
1) echo " overcommit 1 always — every allocation succeeds; the OOM killer is the only limit," ;;
|
||||
2) echo " overcommit 2 strict (ratio ${or_}%) — allocation fails honestly instead of killing later," ;;
|
||||
*) echo " overcommit ${om}" ;;
|
||||
esac
|
||||
[ "$om" != "?" ] && echo " so RSS is the number to trust, not what a process asked for"
|
||||
|
||||
# ── what it says it has ──
|
||||
total=$(mb MemTotal); avail=$(avail_meminfo_mb)
|
||||
swap_total=$(mb SwapTotal); swap_free=$(mb SwapFree)
|
||||
echo
|
||||
echo "memory"
|
||||
echo " total ${total} MB"
|
||||
echo " available ${avail} MB"
|
||||
echo " swap ${swap_total} MB ($(( swap_total - swap_free )) MB used)"
|
||||
if [ "$swap_total" -eq 0 ]; then
|
||||
echo " - no swap: this box has no cushion. It goes from fine to OOM-killed"
|
||||
echo " with nothing in between, which is the abrupt failure you get in a VM."
|
||||
fi
|
||||
|
||||
# postgres puts its shared buffers in /dev/shm. Docker's default is 64 MB,
|
||||
# and the resulting failure names neither shm nor the size.
|
||||
if [ -d /dev/shm ]; then
|
||||
local shm; shm=$(df -Pm /dev/shm 2>/dev/null | awk 'NR==2{print $2}')
|
||||
if [ -n "$shm" ]; then
|
||||
if [ "$shm" -le 64 ]; then
|
||||
echo " ! /dev/shm ${shm} MB — postgres puts shared memory here and 64 MB"
|
||||
echo " is docker's default. Raise it with --shm-size when postgres fails."
|
||||
else
|
||||
echo " /dev/shm ${shm} MB"
|
||||
fi
|
||||
fi
|
||||
fi
|
||||
|
||||
if [ -n "$conf_mb" ]; then
|
||||
# The VM reports a little less than allocated; 15% covers the kernel
|
||||
# without calling every healthy machine a mismatch.
|
||||
if [ "$total" -lt $(( conf_mb * 85 / 100 )) ]; then
|
||||
echo
|
||||
echo "! configured ${conf_mb} MB but booted ${total} MB."
|
||||
echo " The change has not been applied. From a WINDOWS terminal:"
|
||||
echo
|
||||
echo " wsl --shutdown"
|
||||
echo "disk"
|
||||
local d
|
||||
for d in / /tmp /var/lib/docker; do
|
||||
[ -d "$d" ] || continue
|
||||
df -Pm "$d" 2>/dev/null | awk -v p="$d" 'NR==2{printf " %-12s %s MB free of %s MB\n", p, $4, $2}'
|
||||
done
|
||||
|
||||
# kind and Tilt both watch large trees, and the failure mode is silent:
|
||||
# they simply stop noticing file changes. Cheap to report while we are here.
|
||||
local w i
|
||||
w=$(cat /proc/sys/fs/inotify/max_user_watches 2>/dev/null || echo 0)
|
||||
i=$(cat /proc/sys/fs/inotify/max_user_instances 2>/dev/null || echo 0)
|
||||
echo
|
||||
echo " then start the distro again."
|
||||
echo "tooling"
|
||||
echo " inotify watches=$w instances=$i"
|
||||
if [ "$w" -lt 524288 ] || [ "$i" -lt 512 ]; then
|
||||
echo " ! low — anything watching files will silently stop seeing changes"
|
||||
fi
|
||||
|
||||
if ! command -v docker >/dev/null 2>&1; then
|
||||
if [ -S /var/run/docker.sock ]; then
|
||||
echo " docker socket present, no cli"
|
||||
else
|
||||
echo " docker not installed"
|
||||
fi
|
||||
elif docker info >/dev/null 2>&1; then
|
||||
local n
|
||||
n=$(docker ps -q 2>/dev/null | wc -l)
|
||||
echo " docker $(docker version --format '{{.Server.Version}}' 2>/dev/null), ${n} container(s) running"
|
||||
else
|
||||
echo " ! docker cli present but the daemon is unreachable"
|
||||
fi
|
||||
|
||||
# WSL: report the .wslconfig cap and whether it was applied (needs wsl --shutdown).
|
||||
if is_wsl; then
|
||||
local cfg conf conf_mb n
|
||||
cfg=$(wslconfig_path)
|
||||
echo
|
||||
echo "To raise it, add to $cfg on the Windows side:"
|
||||
echo
|
||||
echo "wsl"
|
||||
if [ -z "$cfg" ]; then
|
||||
echo " ! cannot tell which Windows profile owns .wslconfig"
|
||||
else
|
||||
echo " config $cfg"
|
||||
conf=$(configured_memory "$cfg")
|
||||
if [ -n "$conf" ]; then
|
||||
conf_mb=$(to_mb "$conf")
|
||||
echo " configured $conf (${conf_mb} MB), booted ${total} MB"
|
||||
# The VM reports a little less than allocated; 15% covers the
|
||||
# kernel without calling every healthy machine a mismatch.
|
||||
if [ -n "$conf_mb" ] && [ "$total" -lt $(( conf_mb * 85 / 100 )) ]; then
|
||||
echo " ! configured ${conf_mb} MB but booted ${total} MB — not applied yet."
|
||||
echo " From a WINDOWS terminal: wsl --shutdown then start the distro again."
|
||||
fi
|
||||
else
|
||||
echo " configured no memory= set (WSL defaults to 50% of host RAM, or 8 GB,"
|
||||
echo " whichever is less). To raise it, add on the Windows side:"
|
||||
echo " [wsl2]"
|
||||
echo " memory=8GB"
|
||||
echo
|
||||
echo "then, from a WINDOWS terminal: wsl --shutdown"
|
||||
echo " then from a WINDOWS terminal: wsl --shutdown"
|
||||
fi
|
||||
|
||||
local n
|
||||
n=$(ls "$cfg".*.bak 2>/dev/null | wc -l)
|
||||
[ "$n" -gt 0 ] && echo "backups $n (newest: $(ls -t "$cfg".*.bak 2>/dev/null | head -1))"
|
||||
if [ "$n" -gt 0 ]; then
|
||||
echo " backups $n (newest: $(ls -t "$cfg".*.bak 2>/dev/null | head -1))"
|
||||
fi
|
||||
fi
|
||||
else
|
||||
echo "platform native linux"
|
||||
echo "total ${total} MB"
|
||||
echo "available ${avail} MB"
|
||||
echo "swap ${swap_total} MB ($(( swap_total - swap_free )) MB used)"
|
||||
echo
|
||||
echo "No VM allocation to raise here — this is the machine's own memory."
|
||||
echo "If it is tight the levers are freeing something or adding swap."
|
||||
echo " - native linux: no VM allocation to raise. If memory is tight the levers"
|
||||
echo " are freeing something or adding swap."
|
||||
fi
|
||||
|
||||
# Under a fifth left is worth naming wherever you are running.
|
||||
if [ "$avail" -lt $(( total / 5 )) ]; then
|
||||
echo
|
||||
hogs
|
||||
fi
|
||||
echo "effective ceiling $(effective_ceiling_mb) MB"
|
||||
echo " the lowest of MemTotal, the cgroup cap and ulimit -v. What the box"
|
||||
echo " claims. 'push' measures what it will actually hand over."
|
||||
|
||||
[ "$avail" -lt $(( total / 5 )) ] && { echo; hogs; }
|
||||
return 0
|
||||
}
|
||||
|
||||
# ── .wslconfig ─────────────────────────────────────────────────────────────
|
||||
|
||||
require_wsl() {
|
||||
if ! is_wsl; then
|
||||
echo "$1 acts on .wslconfig, which only exists under WSL." >&2
|
||||
echo "This is native Linux — there is no VM allocation to save or roll back." >&2
|
||||
echo "Use 'status' to see what the machine actually has." >&2
|
||||
exit 1
|
||||
fi
|
||||
}
|
||||
|
||||
# backup and restore act on the file, so unlike status they must not guess.
|
||||
wslconfig_required() {
|
||||
local cfg; cfg=$(wslconfig_path)
|
||||
if [ -z "$cfg" ]; then
|
||||
echo "cannot tell which Windows profile owns .wslconfig. Candidates:" >&2
|
||||
ls -d /mnt/c/Users/*/ 2>/dev/null \
|
||||
| grep -viE "/(All Users|Default|Default User|Public)/$" | sed "s/^/ /" >&2
|
||||
exit 1
|
||||
fi
|
||||
echo "$cfg"
|
||||
}
|
||||
|
||||
configured_memory() {
|
||||
[ -r "$1" ] || { echo ""; return; }
|
||||
sed -n 's/^[[:space:]]*memory[[:space:]]*=[[:space:]]*//p' "$1" | tail -1 | tr -d '[:space:]'
|
||||
}
|
||||
|
||||
# "9GB" / "8192MB" / "9G" -> MB, so it can be compared with /proc/meminfo.
|
||||
to_mb() {
|
||||
local v="${1^^}" n
|
||||
n=$(echo "$v" | tr -dc '0-9')
|
||||
[ -n "$n" ] || { echo ""; return; }
|
||||
case "$v" in
|
||||
*GB|*G) echo $(( n * 1024 )) ;;
|
||||
*MB|*M) echo "$n" ;;
|
||||
*) echo $(( n / 1024 / 1024 )) ;;
|
||||
esac
|
||||
}
|
||||
|
||||
backup() {
|
||||
require_wsl backup
|
||||
local cfg dest
|
||||
cfg=$(wslconfig_path)
|
||||
cfg=$(wslconfig_required)
|
||||
[ -r "$cfg" ] || { echo "nothing to back up: $cfg does not exist" >&2; exit 1; }
|
||||
# Timestamped and never overwritten: a backup that can destroy itself on a
|
||||
# second run is not a backup.
|
||||
# Timestamped, never overwritten.
|
||||
dest="${cfg}.$(date +%Y%m%d-%H%M%S).bak"
|
||||
cp "$cfg" "$dest"
|
||||
echo "backed up $dest"
|
||||
@@ -183,7 +398,7 @@ backup() {
|
||||
restore() {
|
||||
require_wsl restore
|
||||
local cfg newest count
|
||||
cfg=$(wslconfig_path)
|
||||
cfg=$(wslconfig_required)
|
||||
newest=$(ls -t "$cfg".*.bak 2>/dev/null | head -1 || true)
|
||||
[ -n "$newest" ] || { echo "no backups found beside $cfg" >&2; exit 1; }
|
||||
|
||||
@@ -191,9 +406,7 @@ restore() {
|
||||
echo " -> $cfg"
|
||||
echo
|
||||
|
||||
# Newest is the right default — undo the last edit — but if you backed up
|
||||
# *after* editing, the state you want is older. Show the rest so a no-op
|
||||
# restore is obviously a no-op rather than a mystery.
|
||||
# Restores the newest; list the others in case an older one is wanted.
|
||||
count=$(ls "$cfg".*.bak 2>/dev/null | wc -l)
|
||||
if [ "$count" -gt 1 ]; then
|
||||
echo "$count backups exist, newest first:"
|
||||
@@ -223,11 +436,249 @@ restore() {
|
||||
echo "restored. From a WINDOWS terminal: wsl --shutdown"
|
||||
}
|
||||
|
||||
# ── push ───────────────────────────────────────────────────────────────────
|
||||
|
||||
STATE=""
|
||||
CHILD=""
|
||||
|
||||
cleanup() {
|
||||
if [ -n "$CHILD" ] && kill -0 "$CHILD" 2>/dev/null; then
|
||||
kill -KILL "$CHILD" 2>/dev/null || true
|
||||
wait "$CHILD" 2>/dev/null || true
|
||||
fi
|
||||
[ -n "$STATE" ] && rm -f "$STATE"
|
||||
return 0
|
||||
}
|
||||
|
||||
# Runs as a child that may be OOM-killed; the parent survives to report.
|
||||
allocator() {
|
||||
# Make this process the preferred OOM victim (raising needs no privilege).
|
||||
echo 1000 > "/proc/$BASHPID/oom_score_adj" 2>/dev/null || true
|
||||
|
||||
local arr=() held=0 i=0 rss swapped avail first_swap=0
|
||||
local bytes=$((STEP_MB * 1024 * 1024))
|
||||
local swap_used_start
|
||||
swap_used_start=$(( $(mb SwapTotal) - $(mb SwapFree) ))
|
||||
|
||||
while :; do
|
||||
# Write straight into the element (one copy, not three) and touch every page.
|
||||
printf -v "arr[$i]" '%*s' "$bytes" ''
|
||||
i=$((i + 1)); held=$((held + STEP_MB))
|
||||
|
||||
rss=$(awk '/^VmRSS:/{print int($2/1024)}' "/proc/$BASHPID/status" 2>/dev/null || echo 0)
|
||||
avail=$(headroom_mb)
|
||||
swapped=$(( $(mb SwapTotal) - $(mb SwapFree) - swap_used_start ))
|
||||
[ "$swapped" -lt 0 ] && swapped=0
|
||||
|
||||
printf '%8s MB held rss %7s MB headroom %7s MB swap +%s MB\n' \
|
||||
"$held" "$rss" "$avail" "$swapped"
|
||||
printf '%s %s %s %s\n' "$held" "$rss" "$avail" "$swapped" >> "$STATE"
|
||||
|
||||
# First swap is reported separately: slow comes before killed.
|
||||
if [ "$swapped" -gt 0 ] && [ "$first_swap" -eq 0 ]; then
|
||||
first_swap=$held
|
||||
echo " - first swap page at ${held} MB — past here it works but crawls"
|
||||
echo "swapat $held" >> "$STATE"
|
||||
fi
|
||||
|
||||
if [ -n "$TO_MB" ] && [ "$held" -ge "$TO_MB" ]; then
|
||||
echo "stop reached-the-cap" >> "$STATE"; return 0
|
||||
fi
|
||||
if [ "$TO_OOM" = no ] && [ "$avail" -lt "$FLOOR_MB" ]; then
|
||||
echo "stop floor" >> "$STATE"; return 0
|
||||
fi
|
||||
done
|
||||
}
|
||||
|
||||
push() {
|
||||
local total ceiling rc=0 last held rss swapat stop
|
||||
total=$(mb MemTotal)
|
||||
ceiling=$(effective_ceiling_mb)
|
||||
|
||||
# Default step: ceiling/64, clamped to 4..256 MB.
|
||||
if [ "$STEP_EXPLICIT" = no ]; then
|
||||
STEP_MB=$(( ceiling / 64 ))
|
||||
[ "$STEP_MB" -lt 4 ] && STEP_MB=4
|
||||
[ "$STEP_MB" -gt 256 ] && STEP_MB=256
|
||||
fi
|
||||
|
||||
# Stop with a cushion: 64 MB under a cgroup cap, 512 MB on a host, or 5% of ceiling if larger.
|
||||
if [ -n "$(cgroup_cap_mb)" ]; then FLOOR_MB=64; else FLOOR_MB=512; fi
|
||||
[ $(( ceiling / 20 )) -gt "$FLOOR_MB" ] && FLOOR_MB=$(( ceiling / 20 ))
|
||||
|
||||
STATE=$(mktemp "${TMPDIR:-/tmp}/rigmini.XXXXXX")
|
||||
trap cleanup EXIT
|
||||
# Ctrl-C kills the child, frees the memory, and still prints the summary.
|
||||
trap 'echo; echo " interrupted"; echo "stop interrupted" >> "$STATE"; [ -n "$CHILD" ] && kill -KILL "$CHILD" 2>/dev/null || true' INT
|
||||
|
||||
echo "push"
|
||||
echo " step ${STEP_MB} MB per allocation, every page touched"
|
||||
echo " ceiling ${ceiling} MB claimed"
|
||||
if [ -n "$TO_MB" ]; then
|
||||
echo " stopping at ${TO_MB} MB (--to)"
|
||||
elif [ "$TO_OOM" = yes ]; then
|
||||
echo " ! stopping only when the kernel stops it (--to-oom)"
|
||||
echo " the allocating child is marked as the preferred OOM victim,"
|
||||
echo " but nothing about an OOM kill is entirely polite. Not on a box"
|
||||
echo " running anything you mind losing."
|
||||
else
|
||||
echo " stopping when headroom drops below ${FLOOR_MB} MB"
|
||||
fi
|
||||
echo
|
||||
|
||||
allocator &
|
||||
CHILD=$!
|
||||
wait "$CHILD" || rc=$?
|
||||
CHILD=""
|
||||
trap - INT
|
||||
|
||||
last=$(grep -E '^[0-9]' "$STATE" 2>/dev/null | tail -1 || true)
|
||||
held=$(echo "$last" | awk '{print $1}')
|
||||
rss=$(echo "$last" | awk '{print $2}')
|
||||
swapat=$(awk '/^swapat/{print $2}' "$STATE" 2>/dev/null | head -1 || true)
|
||||
stop=$(awk '/^stop/{print $2}' "$STATE" 2>/dev/null | head -1 || true)
|
||||
|
||||
echo
|
||||
if [ -z "$held" ]; then
|
||||
echo " ! nothing was allocated. Even one ${STEP_MB} MB chunk failed —"
|
||||
echo " try a smaller --step, or check ulimit -v in 'status'."
|
||||
return 1
|
||||
fi
|
||||
|
||||
echo " reached ${rss:-$held} MB resident"
|
||||
[ -n "$swapat" ] && echo " swapping from ${swapat} MB"
|
||||
|
||||
case "$stop" in
|
||||
reached-the-cap)
|
||||
echo " outcome stopped at the --to cap, not at a limit."
|
||||
echo " The box held ${TO_MB} MB without complaint; there is more." ;;
|
||||
floor)
|
||||
echo " outcome stopped with a cushion intact, by choice."
|
||||
echo " The real ceiling is higher — --to-oom finds it, at the"
|
||||
echo " cost of an actual OOM kill." ;;
|
||||
interrupted)
|
||||
echo " outcome interrupted at ${rss:-$held} MB — where you stopped it,"
|
||||
echo " not where the box did." ;;
|
||||
*)
|
||||
# No stop line means the child did not decide to stop: it was ended.
|
||||
if [ "$rc" -ge 128 ]; then
|
||||
echo " outcome the child was killed (signal $((rc - 128))) at ${rss:-$held} MB."
|
||||
elif [ "$rc" -ne 0 ]; then
|
||||
echo " outcome the allocation failed at ${rss:-$held} MB (exit ${rc})."
|
||||
echo " bash could not get the next chunk — an honest malloc"
|
||||
echo " failure rather than a kill. That is the strict-overcommit"
|
||||
echo " or ulimit path."
|
||||
else
|
||||
echo " outcome ended at ${rss:-$held} MB."
|
||||
fi
|
||||
local ev
|
||||
ev=$(dmesg 2>/dev/null | tail -80 | grep -iE 'oom-kill|killed process' | tail -1 || true)
|
||||
if [ -n "$ev" ]; then
|
||||
echo " kernel ${ev#*] }"
|
||||
else
|
||||
echo " - dmesg is unreadable here (dmesg_restrict, or no privilege),"
|
||||
echo " so the kill cannot be confirmed from this side. The number stands."
|
||||
fi ;;
|
||||
esac
|
||||
|
||||
# Warn about claimed-vs-measured gap only when the box, not us, chose the stop.
|
||||
local got="${rss:-$held}"
|
||||
echo
|
||||
if [ -z "$stop" ] && [ "$got" -lt $(( ceiling * 70 / 100 )) ]; then
|
||||
echo " ! claimed ${ceiling} MB, gave up ${got} MB — under 70% of it."
|
||||
echo " Something is taking the difference. 'status' names the candidates:"
|
||||
echo " a cgroup cap, ulimit -v, or memory already resident."
|
||||
fi
|
||||
return 0
|
||||
}
|
||||
|
||||
# ── all ────────────────────────────────────────────────────────────────────
|
||||
|
||||
all() {
|
||||
status
|
||||
echo
|
||||
echo "────────────────────────────────────────────────────────────"
|
||||
echo
|
||||
push
|
||||
|
||||
local got budget_mb ceiling
|
||||
load_config
|
||||
if [ -n "$BUDGET_GB" ]; then
|
||||
budget_mb=$(( BUDGET_GB * 1024 ))
|
||||
else
|
||||
budget_mb=$(( NODES * NODE_MB ))
|
||||
fi
|
||||
ceiling=$(effective_ceiling_mb)
|
||||
got=$(grep -E '^[0-9]' "$STATE" 2>/dev/null | tail -1 | awk '{print $2}' || true)
|
||||
[ -n "$got" ] || got=0
|
||||
|
||||
echo
|
||||
echo "verdict"
|
||||
if [ -n "$BUDGET_GB" ]; then
|
||||
echo " budget ${budget_mb} MB (--budget)"
|
||||
else
|
||||
# rig's own figure for this profile: nodes times what one node costs.
|
||||
# Addons carry no memory figure in rig yet, so this is the cluster alone
|
||||
# and whatever you deploy comes on top. --budget once you know that too.
|
||||
echo " budget ${budget_mb} MB — profile ${PROFILE_NAME}: ${NODES} node(s) x ${NODE_MB} MB,"
|
||||
echo " the cluster alone; your workload comes on top (--budget GB)"
|
||||
fi
|
||||
echo " measured ${got} MB handed over"
|
||||
|
||||
if [ "$got" -ge "$budget_mb" ]; then
|
||||
echo " fits, with $(( got - budget_mb )) MB spare."
|
||||
if [ "$got" -lt $(( budget_mb * 130 / 100 )) ]; then
|
||||
echo " - under 30% spare is thin once a workload runs on top: memory use"
|
||||
echo " is spiky, and the spikes are what get killed."
|
||||
fi
|
||||
else
|
||||
echo " ! short by $(( budget_mb - got )) MB."
|
||||
if [ "$ceiling" -ge "$budget_mb" ]; then
|
||||
echo " The box CLAIMS enough (${ceiling} MB) but did not deliver it."
|
||||
echo " Free something, or read the caps section again."
|
||||
else
|
||||
echo " The box does not have it to give. A bigger machine, or a profile"
|
||||
echo " with fewer nodes."
|
||||
fi
|
||||
fi
|
||||
return 0
|
||||
}
|
||||
|
||||
# ── main ───────────────────────────────────────────────────────────────────
|
||||
|
||||
parse_flags() {
|
||||
while [ $# -gt 0 ]; do
|
||||
case "$1" in
|
||||
--to) TO_MB=$(( ${2:?--to needs a value in GB} * 1024 )); shift 2 ;;
|
||||
--to-mb) TO_MB="${2:?--to-mb needs a value in MB}"; shift 2 ;;
|
||||
--step) STEP_MB="${2:?--step needs a value in MB}"; STEP_EXPLICIT=yes; shift 2 ;;
|
||||
--to-oom) TO_OOM=yes; shift ;;
|
||||
--budget) BUDGET_GB="${2:?--budget needs a value in GB}"; BUDGET_EXPLICIT=yes; shift 2 ;;
|
||||
*) echo "unknown argument: $1" >&2; exit 1 ;;
|
||||
esac
|
||||
done
|
||||
if [ "$TO_OOM" = yes ] && [ -n "$TO_MB" ]; then
|
||||
echo "--to and --to-oom contradict each other: one stops early, the other" >&2
|
||||
echo "refuses to stop at all. Pick one." >&2
|
||||
exit 1
|
||||
fi
|
||||
return 0
|
||||
}
|
||||
|
||||
require_linux
|
||||
find_cgroup
|
||||
|
||||
cmd="${1:-status}"
|
||||
[ $# -gt 0 ] && shift
|
||||
|
||||
case "${1:-status}" in
|
||||
status) status ;;
|
||||
case "$cmd" in
|
||||
status) parse_flags "$@"; status ;;
|
||||
push) parse_flags "$@"; push ;;
|
||||
all) parse_flags "$@"; all ;;
|
||||
backup) backup ;;
|
||||
restore) restore ;;
|
||||
*) echo "usage: $0 [status|backup|restore]" >&2; exit 1 ;;
|
||||
*) echo "usage: $0 [status|push|all|backup|restore]" >&2
|
||||
echo " push [--to GB] [--to-mb MB] [--step MB] [--to-oom]" >&2
|
||||
echo " all [--budget GB]" >&2
|
||||
exit 1 ;;
|
||||
esac
|
||||
|
||||
@@ -1,316 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Create a disposable Linux environment to validate the installer from a
|
||||
# genuinely clean slate — one that can be thrown away without touching the
|
||||
# environment you actually work in.
|
||||
#
|
||||
# This is the ONLY host-aware file in the tree. Everything else needs just a
|
||||
# Linux with Docker, which is what keeps other host types a later addition
|
||||
# rather than a rewrite.
|
||||
#
|
||||
# On WSL it creates a second distro. There is no .bat and no PowerShell script:
|
||||
# wsl.exe is callable from inside WSL, and wslpath converts the paths it wants.
|
||||
# A machine with no WSL at all needs `wsl --install` run once by hand first —
|
||||
# scripting a reboot-requiring Windows feature install is not worth it.
|
||||
#
|
||||
# Docker: borrowed by default, never installed twice
|
||||
# --------------------------------------------------
|
||||
# WSL2 distros share one kernel and one network stack, so two dockerd instances
|
||||
# contend over docker0 and iptables and can disturb the daemon you depend on.
|
||||
# (That is why Docker Desktop runs one daemon in a dedicated distro and shares
|
||||
# its socket rather than installing one per distro.)
|
||||
#
|
||||
# REUSE_DOCKER=1 (default) borrow the host distro's daemon over /mnt/wsl.
|
||||
# Nothing is installed; nothing can conflict.
|
||||
# Requires `ctrl/dockerhost.sh share` once on the
|
||||
# distro that owns Docker.
|
||||
# REUSE_DOCKER=0 install a second daemon in the new distro. Only
|
||||
# if you specifically want to test a from-scratch
|
||||
# Docker install, and not on a machine you need.
|
||||
#
|
||||
# Borrowing is also the more honest test: rig never installs Docker anyway — it
|
||||
# is the documented prerequisite — so a clean box does not need its own to
|
||||
# exercise everything rig actually does.
|
||||
#
|
||||
# Usage: newbox.sh create | destroy [--purge] | status | shell
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")"
|
||||
|
||||
source ./lib/config.sh
|
||||
load_config
|
||||
|
||||
REPO="$(cd .. && pwd)"
|
||||
|
||||
# The distro is named after this environment, and that derived name is the ONLY
|
||||
# thing this script will ever destroy. See guard_name().
|
||||
BOX="${BOX:-${CLUSTER}box}"
|
||||
BOX_USER="${BOX_USER:-dev}"
|
||||
|
||||
# Borrow the host distro's Docker rather than installing a second daemon.
|
||||
REUSE_DOCKER="${REUSE_DOCKER:-1}"
|
||||
SHARED_SOCK=/mnt/wsl/shared-docker/docker.sock
|
||||
|
||||
WSL_EXE=/mnt/c/Windows/System32/wsl.exe
|
||||
|
||||
# ── host detection ─────────────────────────────────────────────────────────
|
||||
|
||||
require_wsl() {
|
||||
if ! grep -qi microsoft /proc/version 2>/dev/null; then
|
||||
cat >&2 <<'EOF'
|
||||
newbox is WSL-only for now.
|
||||
|
||||
If WSL is not installed, run `wsl --install` from an elevated Windows prompt
|
||||
first — see "Starting from plain Windows" in README.md.
|
||||
|
||||
On native Linux you do not need it: rig already isolates environments by
|
||||
directory (own cluster, context, images and port block), so a second copy in a
|
||||
second directory is the clean slate. To validate the installer itself against a
|
||||
bare system, run ctrl/deps.sh against a stock Debian container instead.
|
||||
EOF
|
||||
exit 1
|
||||
fi
|
||||
if [ ! -x "$WSL_EXE" ]; then
|
||||
echo "wsl.exe not found at $WSL_EXE" >&2
|
||||
exit 1
|
||||
fi
|
||||
}
|
||||
|
||||
wsl_list() { "$WSL_EXE" -l -q 2>/dev/null | tr -d '\0\r'; }
|
||||
box_exists() { wsl_list | grep -qx "$BOX"; }
|
||||
|
||||
# `wsl --unregister` permanently deletes a distro's filesystem. The whole safety
|
||||
# story is this function: only the name derived from this directory can ever be
|
||||
# a target, so a typo or a stray argument cannot destroy the distro you work in.
|
||||
guard_name() {
|
||||
local derived="${CLUSTER}box"
|
||||
if [ "$BOX" != "$derived" ]; then
|
||||
echo "refusing: BOX='$BOX' is not the name derived from this directory ('$derived')." >&2
|
||||
echo "That guard exists because --unregister is irreversible." >&2
|
||||
exit 1
|
||||
fi
|
||||
if [ -z "$CLUSTER" ] || [ "$BOX" = "box" ]; then
|
||||
echo "refusing: empty environment name" >&2
|
||||
exit 1
|
||||
fi
|
||||
}
|
||||
|
||||
# ── create ─────────────────────────────────────────────────────────────────
|
||||
|
||||
rootfs_path() {
|
||||
local win_home; win_home=$(wslpath "$("$WSL_EXE" -d "$(wsl_list | head -1)" -e printf '%s' "$USERPROFILE" 2>/dev/null || true)" 2>/dev/null || true)
|
||||
# Simpler and reliable: use the current user's Windows home via /mnt/c.
|
||||
ls -d /mnt/c/Users/*/ 2>/dev/null | grep -viE '/(All Users|Default|Default User|Public)/$' | head -1
|
||||
}
|
||||
|
||||
build_rootfs() {
|
||||
local tar="$1"
|
||||
if [ -f "$tar" ]; then
|
||||
echo " rootfs cached: $(basename "$tar")"
|
||||
return
|
||||
fi
|
||||
echo " exporting a stock Debian rootfs (cached for next time)"
|
||||
local cid; cid=$(docker create debian:trixie-slim)
|
||||
docker export "$cid" > "$tar"
|
||||
docker rm -f "$cid" >/dev/null
|
||||
}
|
||||
|
||||
provision() {
|
||||
echo " provisioning (root)"
|
||||
local hosts_block
|
||||
hosts_block=$(CLUSTER="$CLUSTER" HTTP_PORT="$HTTP_PORT" \
|
||||
envsubst < ./hosts.tmpl 2>/dev/null || sed "s/\${CLUSTER}/$CLUSTER/g" ./hosts.tmpl)
|
||||
|
||||
# Piped as stdin rather than a second script file, the same shape as any
|
||||
# remote provisioning heredoc. Everything here is idempotent so a failed run
|
||||
# can simply be repeated.
|
||||
"$WSL_EXE" -d "$BOX" -u root -- bash -s <<PROVISION
|
||||
set -euo pipefail
|
||||
|
||||
export DEBIAN_FRONTEND=noninteractive
|
||||
apt-get update -qq
|
||||
apt-get install -y -qq ca-certificates curl gnupg sudo >/dev/null
|
||||
|
||||
if [ "$REUSE_DOCKER" = "1" ]; then
|
||||
# Borrow the host distro's daemon: CLI only, no dockerd, nothing to
|
||||
# conflict with. The GID must match the owner's or the shared socket is
|
||||
# unreadable here even though it is visible.
|
||||
install -m 0755 -d /etc/apt/keyrings
|
||||
if [ ! -f /etc/apt/keyrings/docker.asc ]; then
|
||||
curl -fsSL https://download.docker.com/linux/debian/gpg -o /etc/apt/keyrings/docker.asc
|
||||
chmod a+r /etc/apt/keyrings/docker.asc
|
||||
fi
|
||||
echo "deb [arch=\$(dpkg --print-architecture) signed-by=/etc/apt/keyrings/docker.asc] https://download.docker.com/linux/debian \$(. /etc/os-release && echo \$VERSION_CODENAME) stable" \
|
||||
> /etc/apt/sources.list.d/docker.list
|
||||
apt-get update -qq
|
||||
apt-get install -y -qq docker-ce-cli >/dev/null
|
||||
|
||||
echo "export DOCKER_HOST=unix://$SHARED_SOCK" > /etc/profile.d/rig-docker-host.sh
|
||||
|
||||
if [ -f /mnt/wsl/shared-docker/OWNER ]; then
|
||||
gid=\$(awk '/docker gid:/ {print \$3}' /mnt/wsl/shared-docker/OWNER)
|
||||
if [ -n "\$gid" ]; then
|
||||
getent group docker >/dev/null && groupmod -g "\$gid" docker || groupadd -g "\$gid" docker
|
||||
fi
|
||||
fi
|
||||
else
|
||||
# A second daemon. Only when deliberately testing a from-scratch install.
|
||||
install -m 0755 -d /etc/apt/keyrings
|
||||
if [ ! -f /etc/apt/keyrings/docker.asc ]; then
|
||||
curl -fsSL https://download.docker.com/linux/debian/gpg -o /etc/apt/keyrings/docker.asc
|
||||
chmod a+r /etc/apt/keyrings/docker.asc
|
||||
fi
|
||||
echo "deb [arch=\$(dpkg --print-architecture) signed-by=/etc/apt/keyrings/docker.asc] https://download.docker.com/linux/debian \$(. /etc/os-release && echo \$VERSION_CODENAME) stable" \
|
||||
> /etc/apt/sources.list.d/docker.list
|
||||
apt-get update -qq
|
||||
apt-get install -y -qq docker-ce docker-ce-cli containerd.io >/dev/null
|
||||
fi
|
||||
|
||||
id -u "$BOX_USER" >/dev/null 2>&1 || useradd -m -s /bin/bash "$BOX_USER"
|
||||
usermod -aG sudo,docker "$BOX_USER"
|
||||
echo "$BOX_USER ALL=(ALL) NOPASSWD:ALL" > /etc/sudoers.d/90-$BOX_USER
|
||||
chmod 0440 /etc/sudoers.d/90-$BOX_USER
|
||||
|
||||
# systemd is off by default in WSL, and Docker needs it. Takes effect on the
|
||||
# next start of this distro, which is why create() terminates it below.
|
||||
cat > /etc/wsl.conf <<WSLCONF
|
||||
[boot]
|
||||
systemd=true
|
||||
|
||||
[user]
|
||||
default=$BOX_USER
|
||||
WSLCONF
|
||||
|
||||
# The default inotify limits are low enough that file watching silently stops
|
||||
# working — no error, changes just stop being noticed. Fix it before it bites.
|
||||
cat > /etc/sysctl.d/99-rig.conf <<SYSCTL
|
||||
fs.inotify.max_user_watches=524288
|
||||
fs.inotify.max_user_instances=512
|
||||
SYSCTL
|
||||
|
||||
if ! grep -q 'rig environment' /etc/hosts 2>/dev/null; then
|
||||
{ echo ""; echo "# rig environment"; cat <<'HOSTS'
|
||||
$hosts_block
|
||||
HOSTS
|
||||
} >> /etc/hosts
|
||||
fi
|
||||
|
||||
touch /etc/rig-provisioned
|
||||
PROVISION
|
||||
}
|
||||
|
||||
create() {
|
||||
require_wsl
|
||||
guard_name
|
||||
|
||||
local winhome; winhome=$(rootfs_path)
|
||||
[ -n "$winhome" ] || { echo "could not locate the Windows user directory" >&2; exit 1; }
|
||||
local tar="${winhome}rig-rootfs.tar"
|
||||
local installdir="${winhome}WSL/${BOX}"
|
||||
|
||||
echo "creating '$BOX'"
|
||||
if [ "$REUSE_DOCKER" = "1" ]; then
|
||||
echo " docker: borrowing the host distro's daemon (nothing installed)"
|
||||
if [ ! -S "$SHARED_SOCK" ]; then
|
||||
echo
|
||||
echo " No shared socket yet. In the distro that owns Docker, run once:"
|
||||
echo " sudo bash ctrl/dockerhost.sh share"
|
||||
echo " That adds one systemd drop-in and nothing else; undo with 'unshare'."
|
||||
echo " Continuing — the box will be created, but Docker won't work in it"
|
||||
echo " until you do that."
|
||||
fi
|
||||
else
|
||||
echo
|
||||
echo " REUSE_DOCKER=0: installing a SECOND Docker daemon."
|
||||
echo " WSL distros share a network stack, so this can disturb Docker in"
|
||||
echo " the distro you work in. Ctrl-C now if that is a bad trade today."
|
||||
echo
|
||||
sleep 4
|
||||
fi
|
||||
echo
|
||||
|
||||
if box_exists; then
|
||||
echo " distro already registered"
|
||||
else
|
||||
build_rootfs "$tar"
|
||||
mkdir -p "$installdir"
|
||||
"$WSL_EXE" --import "$BOX" "$(wslpath -w "$installdir")" "$(wslpath -w "$tar")" --version 2
|
||||
fi
|
||||
|
||||
# Resumable: a partially-created box is finished rather than restarted.
|
||||
if "$WSL_EXE" -d "$BOX" -u root -- test -f /etc/rig-provisioned 2>/dev/null; then
|
||||
echo " already provisioned"
|
||||
else
|
||||
provision
|
||||
echo " restarting the distro so systemd and group membership apply"
|
||||
"$WSL_EXE" --terminate "$BOX" # ONLY this distro; never --shutdown
|
||||
fi
|
||||
|
||||
echo " copying rig in"
|
||||
tar c -C "$REPO" --exclude=def --exclude=.git --exclude=ctrl/.env . \
|
||||
| "$WSL_EXE" -d "$BOX" -u "$BOX_USER" -- bash -lc "mkdir -p ~/rig && tar x -C ~/rig"
|
||||
|
||||
echo
|
||||
echo " docker: $("$WSL_EXE" -d "$BOX" -u "$BOX_USER" -- bash -lc 'systemctl is-active docker 2>/dev/null || echo inactive')"
|
||||
echo
|
||||
echo "next:"
|
||||
echo " make newbox shell # a shell inside it"
|
||||
echo " then: cd ~/rig && make check && make deps && make cluster up"
|
||||
echo
|
||||
echo "For a browser on Windows to resolve the hostnames, paste this into"
|
||||
echo "C:\\Windows\\System32\\drivers\\etc\\hosts (it has no wildcard support):"
|
||||
CLUSTER="$CLUSTER" envsubst < ./hosts.tmpl 2>/dev/null | grep -v '^#' | grep -v '^$' | sed 's/^/ /'
|
||||
}
|
||||
|
||||
# ── the rest ───────────────────────────────────────────────────────────────
|
||||
|
||||
destroy() {
|
||||
require_wsl
|
||||
guard_name
|
||||
|
||||
if ! box_exists; then
|
||||
echo "no distro '$BOX' to remove"
|
||||
else
|
||||
echo "about to PERMANENTLY delete the distro '$BOX' and its filesystem."
|
||||
"$WSL_EXE" --terminate "$BOX" 2>/dev/null || true
|
||||
"$WSL_EXE" --unregister "$BOX"
|
||||
echo " unregistered"
|
||||
fi
|
||||
|
||||
local winhome; winhome=$(rootfs_path)
|
||||
rm -rf "${winhome}WSL/${BOX}" 2>/dev/null || true
|
||||
|
||||
if [ "${1:-}" = "--purge" ]; then
|
||||
rm -f "${winhome}rig-rootfs.tar"
|
||||
echo " cached rootfs removed"
|
||||
fi
|
||||
}
|
||||
|
||||
status() {
|
||||
require_wsl
|
||||
echo "environment $CLUSTER"
|
||||
echo "distro $BOX"
|
||||
if box_exists; then
|
||||
echo "registered yes"
|
||||
echo "provisioned $("$WSL_EXE" -d "$BOX" -u root -- test -f /etc/rig-provisioned 2>/dev/null && echo yes || echo no)"
|
||||
echo "docker $("$WSL_EXE" -d "$BOX" -u root -- bash -lc 'systemctl is-active docker 2>/dev/null' || echo unknown)"
|
||||
echo "rig copied $("$WSL_EXE" -d "$BOX" -u "$BOX_USER" -- bash -lc 'test -f ~/rig/Makefile && echo yes || echo no' 2>/dev/null)"
|
||||
else
|
||||
echo "registered no"
|
||||
fi
|
||||
echo
|
||||
echo "all distros (this one is never touched unless it is '$BOX'):"
|
||||
wsl_list | sed 's/^/ /'
|
||||
}
|
||||
|
||||
shell() {
|
||||
require_wsl
|
||||
box_exists || { echo "no distro '$BOX' — run 'make newbox' first" >&2; exit 1; }
|
||||
"$WSL_EXE" -d "$BOX" -u "$BOX_USER" --cd '~'
|
||||
}
|
||||
|
||||
case "${1:-status}" in
|
||||
create) create ;;
|
||||
destroy) shift; destroy "${1:-}" ;;
|
||||
status) status ;;
|
||||
shell) shell ;;
|
||||
*) echo "usage: $0 [create|destroy [--purge]|status|shell]" >&2; exit 1 ;;
|
||||
esac
|
||||
@@ -1,57 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Do the standalone scripts still install what rig pins?
|
||||
#
|
||||
# standalone/rigdeps.sh carries its toolchain pins inline, because it exists for
|
||||
# a machine that will never have ctrl/versions.env. That makes two copies of the
|
||||
# same versions and checksums, and two copies drift the day one is edited and
|
||||
# the other forgotten. This is the check that notices.
|
||||
#
|
||||
# ctrl/versions.env is the source of truth. Only the keys rigdeps.sh itself
|
||||
# defines are compared: versions.env also pins addon images (cert-manager,
|
||||
# metallb, metrics-server) that rigdeps.sh never installs, and demanding those
|
||||
# would make this fail forever for no reason.
|
||||
#
|
||||
# Exits non-zero on any mismatch — unlike the host checks, this one is a test.
|
||||
#
|
||||
# Usage: pins.sh
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")"
|
||||
|
||||
SOURCE=./versions.env
|
||||
COPY=../standalone/rigdeps.sh
|
||||
|
||||
[ -r "$COPY" ] || { echo "no $COPY to compare" >&2; exit 1; }
|
||||
|
||||
# KEY=value for the pin keys a file defines, quotes stripped. awk rather than a
|
||||
# grep regex, which is not the same program everywhere.
|
||||
pins() {
|
||||
awk -F= '/^[A-Z_]+_(VERSION|SHA256)=/ {
|
||||
v = substr($0, index($0, "=") + 1); gsub(/^["\x27]|["\x27]$/, "", v)
|
||||
print $1 "=" v }' "$1"
|
||||
}
|
||||
|
||||
echo "pins: standalone/rigdeps.sh against ctrl/versions.env"
|
||||
bad=0
|
||||
while IFS='=' read -r key copy_val; do
|
||||
[ -n "$key" ] || continue
|
||||
src_val=$(pins "$SOURCE" | sed -n "s/^${key}=//p" | head -1)
|
||||
if [ -z "$src_val" ]; then
|
||||
printf " ! %-16s in rigdeps.sh but not in versions.env\n" "$key"
|
||||
bad=1
|
||||
elif [ "$src_val" = "$copy_val" ]; then
|
||||
printf " %-16s %s\n" "$key" "$( [ ${#src_val} -gt 20 ] && echo "${src_val:0:12}…" || echo "$src_val" )"
|
||||
else
|
||||
printf " ! %-16s versions.env %s\n" "$key" "$src_val"
|
||||
printf " %-16s rigdeps.sh %s\n" "" "$copy_val"
|
||||
bad=1
|
||||
fi
|
||||
done < <(pins "$COPY")
|
||||
|
||||
echo
|
||||
if [ "$bad" -eq 0 ]; then
|
||||
echo "in step — rigdeps.sh installs exactly what rig pins."
|
||||
else
|
||||
echo "DRIFT. versions.env is the source of truth: copy the differing lines from it"
|
||||
echo "into standalone/rigdeps.sh, taking checksums from the publisher's release list."
|
||||
exit 1
|
||||
fi
|
||||
@@ -1,25 +1,8 @@
|
||||
#!/usr/bin/env bash
|
||||
# Give each environment its own block of host ports.
|
||||
#
|
||||
# New versions of a system mean new clusters on ONE machine, not new machines.
|
||||
# Cluster name, kubectl context, registry container and image tag already derive
|
||||
# from the directory name, so two copies never collide there — but host ports are
|
||||
# a single shared namespace and would.
|
||||
#
|
||||
# The block is derived from the directory name: stateless, stable, and requiring
|
||||
# no coordination between copies that know nothing about each other.
|
||||
#
|
||||
# base = 20000 + (hash(slug) % 200) * 10
|
||||
# +0 HTTP +1 HTTPS +2 TILT +3 REGISTRY (+4..9 reserved)
|
||||
#
|
||||
# 20000+ deliberately avoids the ports something is already likely to hold: 80,
|
||||
# 443, 3000, 5432, 8000, 8080.
|
||||
#
|
||||
# Derivation is a default, not a decision. On first use the resolved block is
|
||||
# written into ctrl/.env, so it becomes pinned, visible and editable rather than
|
||||
# a number that appears from nowhere. Anything already in ctrl/.env wins.
|
||||
#
|
||||
# Usage: ports.sh show | derive | persist
|
||||
# Give each environment its own block of host ports, derived from the directory name.
|
||||
# base = 20000 + (hash(slug) % 200) * 10; +0 HTTP +1 HTTPS +2 TILT +3 REGISTRY
|
||||
# Usage: ports.sh show | active | derive | persist
|
||||
# Notes: docs/notes/ports.md
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")"
|
||||
|
||||
@@ -38,6 +21,22 @@ derive() {
|
||||
DERIVED_REGISTRY=$((base + 3))
|
||||
}
|
||||
|
||||
# Resolved facts for consumers outside bash, space-separated, positional:
|
||||
# CLUSTER KUBECONTEXT HTTP HTTPS TILT REGISTRY MANIFESTS_DIR OVERLAY_DIR
|
||||
# The two paths are absolute, or - when there is none. Read this, not `derive`.
|
||||
active() {
|
||||
load_config
|
||||
local m="-" o="-"
|
||||
if [ -n "$MANIFESTS_DIR" ]; then m=$(_abs_from_ctrl "$MANIFESTS_DIR"); fi
|
||||
if [ -n "$OVERLAY_DIR" ]; then o=$(_abs_from_ctrl "$OVERLAY_DIR"); fi
|
||||
case "$m$o" in
|
||||
*[[:space:]]*)
|
||||
echo "a path here holds whitespace, and these facts are split on spaces: $m $o" >&2
|
||||
exit 1 ;;
|
||||
esac
|
||||
echo "$CLUSTER $KUBECONTEXT $HTTP_PORT $HTTPS_PORT $TILT_PORT $REGISTRY_PORT $m $o"
|
||||
}
|
||||
|
||||
show() {
|
||||
derive
|
||||
echo "environment $CLUSTER"
|
||||
@@ -64,6 +63,13 @@ _row() {
|
||||
# rewritten — an override stays an override.
|
||||
persist() {
|
||||
derive
|
||||
# ctrl/.env belongs to this rig, not to an overlay: a pin written now would
|
||||
# follow every overlay this rig later runs, and two of them would collide.
|
||||
if [ -n "${OVERLAY:-}" ]; then
|
||||
echo "not pinning: OVERLAY is set, and ctrl/.env would carry this block to every overlay" >&2
|
||||
echo " its ports stay derived from its folder name ($CLUSTER): $DERIVED_HTTP-$DERIVED_REGISTRY" >&2
|
||||
exit 1
|
||||
fi
|
||||
[ -f ./.env ] || cp ./.env.example ./.env
|
||||
|
||||
local wrote=0 key val
|
||||
@@ -98,6 +104,7 @@ persist() {
|
||||
case "${1:-show}" in
|
||||
show) show ;;
|
||||
derive) derive; echo "$DERIVED_HTTP $DERIVED_HTTPS $DERIVED_TILT $DERIVED_REGISTRY" ;;
|
||||
active) active ;;
|
||||
persist) persist ;;
|
||||
*) echo "usage: $0 [show|derive|persist]" >&2; exit 1 ;;
|
||||
*) echo "usage: $0 [show|active|derive|persist]" >&2; exit 1 ;;
|
||||
esac
|
||||
|
||||
@@ -1,28 +1,8 @@
|
||||
#!/usr/bin/env bash
|
||||
# Registry plumbing. THIS is the seam — not a tool.
|
||||
#
|
||||
# Four modes, selected by REGISTRY_MODE in the active profile:
|
||||
#
|
||||
# none Tilt builds straight into the node. No registry at all — and so no
|
||||
# guard against an outward push: an unqualified image name means
|
||||
# docker.io/library/<name>, and only Tilt's kind detection stands
|
||||
# between that and a real push. Throwaway use only; every profile
|
||||
# here now defaults to `local` instead.
|
||||
# local a registry:2 container wired into the cluster.
|
||||
# mirror the same container, but configured as a pull-through CACHE of the
|
||||
# corporate registry. What a locked-down client actually looks like:
|
||||
# images originate from corp, you don't hammer it, and you keep
|
||||
# working when the VPN drops.
|
||||
# remote no local container; pull straight from the corporate registry using
|
||||
# an imagePullSecret.
|
||||
#
|
||||
# Deliberately a script rather than a tool. ctlptl collapses the `local` wiring
|
||||
# into one line, but its Registry spec only accepts name/port/image/listenAddress
|
||||
# — there is no way to set REGISTRY_PROXY_REMOTEURL, so it cannot express
|
||||
# `mirror` at all. Keeping the seam here is what keeps the corporate registry
|
||||
# swappable.
|
||||
#
|
||||
# Registry plumbing: REGISTRY_MODE none | local | mirror | remote. A script, not ctlptl,
|
||||
# because ctlptl cannot express `mirror`.
|
||||
# Usage: registry.sh up | down | status
|
||||
# Notes: docs/notes/registry.md
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")"
|
||||
|
||||
@@ -34,17 +14,7 @@ REG_PORT="${REGISTRY_PORT:-5005}"
|
||||
K="kubectl --context ${KUBECONTEXT}"
|
||||
|
||||
# ── CA trust ───────────────────────────────────────────────────────────────
|
||||
# A corporate registry is almost always fronted by an internal CA, and trust has
|
||||
# to reach three separate places. Nothing does this for you, and the symptom when
|
||||
# it's missing is an opaque:
|
||||
# x509: certificate signed by unknown authority
|
||||
#
|
||||
# 1. the host docker daemon — /etc/docker/certs.d/<host>/ca.crt (needs root)
|
||||
# 2. every kind node's containerd — nodes do NOT inherit host trust
|
||||
# 3. anything doing HTTPS from inside the cluster, in its own trust store
|
||||
#
|
||||
# We handle (2) here because it's ours to handle. (1) is reported by check.sh
|
||||
# since it needs root. (3) belongs to the workload.
|
||||
# Copy REGISTRY_CA_FILE into every kind node's trust store (nodes don't inherit host trust).
|
||||
install_ca_into_nodes() {
|
||||
[ -n "${REGISTRY_CA_FILE:-}" ] || return 0
|
||||
|
||||
|
||||
375
rig/ctrl/selftest.sh
Executable file
375
rig/ctrl/selftest.sh
Executable file
@@ -0,0 +1,375 @@
|
||||
#!/usr/bin/env bash
|
||||
# What rig has settled, written down as assertions: one decision per check.
|
||||
# No cluster, docker or network; exits 1 on failure (unlike `make check`).
|
||||
# Usage: make selftest (or: bash ctrl/selftest.sh)
|
||||
# Notes: docs/notes/selftest.md
|
||||
set -uo pipefail # NOT -e: one failing check must not abort the rest
|
||||
cd "$(dirname "$0")"
|
||||
|
||||
source ./lib/config.sh
|
||||
|
||||
rc=0
|
||||
passed=0
|
||||
|
||||
check() { # name, expected, actual
|
||||
if [ "$2" = "$3" ]; then
|
||||
printf ' ok %s\n' "$1"
|
||||
passed=$((passed + 1))
|
||||
else
|
||||
printf ' FAIL %s\n expected: %s\n got: %s\n' "$1" "$2" "$3"
|
||||
rc=1
|
||||
fi
|
||||
}
|
||||
|
||||
note() { printf '\n%s\n' "$1"; }
|
||||
|
||||
# A scratch copy of rig for a check to change freely. local/ (overlays, possibly
|
||||
# someone else's) and def/ (scratch) never ride along, and neither do this
|
||||
# machine's PROFILE/OVERLAY/CLUSTER choices: a check sets what it tests.
|
||||
copy_rig() { # dest-dir
|
||||
mkdir -p "$1"
|
||||
tar -C .. --exclude=./local --exclude=./def -cf - . | tar -C "$1" -xf -
|
||||
if [ -f "$1/ctrl/.env" ]; then
|
||||
sed -i '/^PROFILE=/d; /^OVERLAY=/d; /^CLUSTER=/d; /^MANIFESTS_DIR=/d' "$1/ctrl/.env"
|
||||
fi
|
||||
}
|
||||
|
||||
# Resolve one key the way every rig script does, in a clean shell so the
|
||||
# caller's exported value is the only thing in play.
|
||||
resolved() {
|
||||
bash -c 'source ./lib/config.sh; load_config >/dev/null 2>&1; printf "%s" "${!1}"' _ "$1"
|
||||
}
|
||||
|
||||
|
||||
note "rig needs no profile"
|
||||
# No env.d/ must still resolve and generate a kit; an unknown profile stays an error.
|
||||
NP="$(mktemp -d)"
|
||||
copy_rig "$NP/rig"; rm -rf "$NP/rig/ctrl/env.d"
|
||||
check "no env.d: config resolves" "default" \
|
||||
"$(cd "$NP/rig/ctrl" && bash -c 'source ./lib/config.sh; load_config >/dev/null && echo "$PROFILE_NAME"' 2>&1)"
|
||||
check "no env.d: the k8s version comes from the pins" "yes" \
|
||||
"$(cd "$NP/rig/ctrl" && bash -c 'source ./lib/config.sh; load_config >/dev/null && [ -n "$NODE_IMAGE" ] && echo yes' 2>&1)"
|
||||
check "no env.d: ports.sh active works" "8" \
|
||||
"$(cd "$NP/rig/ctrl" && bash ports.sh active 2>/dev/null | wc -w)"
|
||||
check "no env.d: a kit is generated for the defaults" "yes" \
|
||||
"$( (cd "$NP/rig/ctrl" && rm -rf ../standalone/*/ && bash standalone.sh write >/dev/null 2>&1) && [ -f "$NP/rig/standalone/default/rigdeps.sh" ] && echo yes || echo no)"
|
||||
check "a profile that does not exist is still an error" "yes" \
|
||||
"$( (cd "$NP/rig/ctrl" && PROFILE=no-such-profile bash -c 'source ./lib/config.sh; load_config' >/dev/null 2>&1) && echo no || echo yes)"
|
||||
rm -rf "$NP"
|
||||
|
||||
|
||||
note "the ports.sh active contract"
|
||||
# ports.sh active is read positionally by the Makefile and Tiltfile: pin field count and order.
|
||||
FACTS="$(bash ports.sh active)"
|
||||
check "active: exactly 8 fields" "8" "$(printf '%s' "$FACTS" | wc -w)"
|
||||
read -r F_CLUSTER F_CTX F_HTTP F_HTTPS F_TILT F_REG F_MANIFESTS F_OVERLAY <<< "$FACTS"
|
||||
check "active: field 2 is kind-<cluster>" "kind-$F_CLUSTER" "$F_CTX"
|
||||
check "active: fields 3-6 are numeric" "yes" \
|
||||
"$([[ "$F_HTTP$F_HTTPS$F_TILT$F_REG" =~ ^[0-9]+$ ]] && echo yes || echo no)"
|
||||
# Absolute, or - when there is none: an empty field would shift every later one.
|
||||
check "active: fields 7-8 are absolute paths or -" "yes" \
|
||||
"$(for f in "$F_MANIFESTS" "$F_OVERLAY"; do case "$f" in -|/*) ;; *) echo no; exit; esac; done; echo yes)"
|
||||
# derive answers a different question and must keep its own shape: it reports
|
||||
# what the directory name implies, ignoring ctrl/.env, so nothing should
|
||||
# configure itself from it.
|
||||
check "derive: still 4 fields, not 7" "4" "$(bash ports.sh derive | wc -w)"
|
||||
|
||||
|
||||
note "the caller's env beats the files"
|
||||
# Every key in CONFIG_OVERRIDABLE must lose to the caller's env; the loop follows the list.
|
||||
test_value() {
|
||||
case "$1" in
|
||||
# Picked from what exists, never named: rig must not need any particular
|
||||
# profile, template or pinned version to be present for this to run.
|
||||
PROFILE) config_profiles | head -1 ;;
|
||||
K8S_VERSION) (set -a; source ./versions.env; compgen -v NODE_IMAGE_v | sort -V | head -1 | sed 's/^NODE_IMAGE_//') ;;
|
||||
# An absolute path, as a project passing its own file does. Never equal
|
||||
# to the default, so the check cannot pass by accident.
|
||||
KIND_CONFIG) echo "$PWD/k8s/kind-config.yaml.tpl" ;;
|
||||
*_PORT) echo "19999" ;;
|
||||
CLUSTER) echo "selftest-name" ;;
|
||||
# Named folders must exist, and must not be the default.
|
||||
MANIFESTS_DIR) echo "examples/starter/k8s/base" ;;
|
||||
OVERLAY) echo "examples/data" ;;
|
||||
ADDONS) echo "metallb" ;;
|
||||
*) echo "selftest-sentinel" ;;
|
||||
esac
|
||||
}
|
||||
for key in $CONFIG_OVERRIDABLE; do
|
||||
[ -n "$key" ] || continue
|
||||
want="$(test_value "$key")"
|
||||
if [ -z "$want" ]; then
|
||||
check "precedence: $key has a test value" "yes" "no — add one to test_value()"
|
||||
continue
|
||||
fi
|
||||
got="$(export "$key=$want"; resolved "$key")"
|
||||
check "precedence: caller's $key wins" "$want" "$got"
|
||||
done
|
||||
|
||||
|
||||
note "one derivation, not three"
|
||||
# The Makefile must take context/port from ports.sh active, checked on real `make -n` output.
|
||||
# --no-print-directory + grep, not tail -1: under `make selftest` this is a recursive make.
|
||||
MK="$(cd .. && make --no-print-directory -n tilt 2>/dev/null | grep -m1 'tilt ')"
|
||||
check "Makefile: --context comes from active" "$F_CTX" \
|
||||
"$(printf '%s' "$MK" | sed -n 's/.*--context \([^ ]*\).*/\1/p')"
|
||||
check "Makefile: --port comes from active" "$F_TILT" \
|
||||
"$(printf '%s' "$MK" | sed -n 's/.*--port \([^ ]*\).*/\1/p')"
|
||||
|
||||
|
||||
note "identity follows the folder, safely"
|
||||
# The cluster name is the folder name made a DNS label, derived only in lib/config.sh.
|
||||
TMP="$(mktemp -d)"
|
||||
trap 'rm -rf "$TMP"' EXIT
|
||||
mkdir -p "$TMP/My_Proj"
|
||||
cp -r . "$TMP/My_Proj/ctrl"
|
||||
# A pinned CLUSTER in .env would be an override, not a derivation, and this
|
||||
# check is about the derivation. (An OVERLAY would be another derivation.)
|
||||
sed -i '/^CLUSTER=/d; /^OVERLAY=/d' "$TMP/My_Proj/ctrl/.env" 2>/dev/null
|
||||
COPY="$(cd "$TMP/My_Proj/ctrl" && bash ports.sh active)"
|
||||
check "a dir named My_Proj derives a DNS label" "my-proj" "$(awk '{print $1}' <<< "$COPY")"
|
||||
check "and a context to match" "kind-my-proj" "$(awk '{print $2}' <<< "$COPY")"
|
||||
check "a renamed copy gets a DIFFERENT block" "different" \
|
||||
"$([ "$(awk '{print $3}' <<< "$COPY")" != "$F_HTTP" ] && echo different || echo COLLIDES)"
|
||||
|
||||
|
||||
note "ports are stable across versions"
|
||||
# Ports are derived, never stored: a changed derivation moves every existing env's ports.
|
||||
check "derive_port_base rig" "20310" "$(derive_port_base rig)"
|
||||
check "derive_port_base foo" "21690" "$(derive_port_base foo)"
|
||||
check "derive_port_base my-proj" "21030" "$(derive_port_base my-proj)"
|
||||
|
||||
|
||||
note "rig stays standalone"
|
||||
# rig must be copyable out of its host project: no references to the host.
|
||||
# The pattern is assembled from fragments so this file does not match itself.
|
||||
# The host project's word for a backing service counts too: rig described its
|
||||
# workload addons with it until they left. local/ holds overlays, which may say anything.
|
||||
HOST_PAT="$(printf '%s' 'sole' 'print' '|\b' 'sp' 'r\b' '|' 'cab' 'inet')"
|
||||
check "no host-project references" "0" \
|
||||
"$(cd .. && grep -rIl -iE "$HOST_PAT" . --exclude-dir=def --exclude-dir=local 2>/dev/null | wc -l)"
|
||||
|
||||
|
||||
note "what runs is an overlay; rig only reads it"
|
||||
# docs/notes/overlay.md. Every check runs in a scratch copy with its own overlay.
|
||||
OV="$TMP/overlay-proof"; copy_rig "$OV/rig"
|
||||
OVR="$OV/rig"
|
||||
mkdir -p "$OVR/local/My_Env/addons" "$OVR/local/My_Env/k8s/prod" "$OVR/ctrl/env.d"
|
||||
printf 'ADDONS="from-profile"\nDATA_NAMESPACE=from-profile\n' > "$OVR/ctrl/env.d/selftest.env"
|
||||
cat > "$OVR/local/My_Env/rig.env" <<'EOF'
|
||||
ADDONS="metallb"
|
||||
DATA_NAMESPACE=from-overlay
|
||||
MANIFESTS_DIR=k8s/prod
|
||||
SELFTEST_SENTINEL=selftest-overlay-sentinel
|
||||
EOF
|
||||
printf 'resources: []\n' > "$OVR/local/My_Env/k8s/prod/kustomization.yaml"
|
||||
printf '#!/usr/bin/env bash\necho "overlay-metallb from $PWD with ${RIG_CTRL:-no RIG_CTRL}"\n' \
|
||||
> "$OVR/local/My_Env/addons/metallb.sh"
|
||||
in_ov() { (cd "$OVR/ctrl" && "$@"); }
|
||||
ov_key() { # key [env assignments...]
|
||||
local k="$1"; shift
|
||||
in_ov env "$@" bash -c 'source ./lib/config.sh; load_config >/dev/null 2>&1; printf "%s" "${!1}"' _ "$k"
|
||||
}
|
||||
|
||||
# With nothing named, rig behaves as it did before overlays: same name, ports, addons, nodes.
|
||||
check "no overlay: the cluster, ports and addons of before" "rig kind-rig 20310 20311 20312 20313" \
|
||||
"$(in_ov bash ports.sh active | awk '{print $1, $2, $3, $4, $5, $6}')"
|
||||
check "no overlay: no addons, one node, rig's own kind config" "|1|./k8s/kind-config.yaml.tpl" \
|
||||
"$(ov_key ADDONS)|$(ov_key NODES)|$(ov_key KIND_CONFIG)"
|
||||
|
||||
# The overlay's rig.env sits between the profile and ctrl/.env; the caller beats all.
|
||||
check "rig.env beats the profile" "from-overlay" \
|
||||
"$(ov_key DATA_NAMESPACE PROFILE=selftest OVERLAY=local/My_Env)"
|
||||
echo 'DATA_NAMESPACE=from-dotenv' >> "$OVR/ctrl/.env"
|
||||
check "ctrl/.env beats rig.env" "from-dotenv" \
|
||||
"$(ov_key DATA_NAMESPACE PROFILE=selftest OVERLAY=local/My_Env)"
|
||||
sed -i '/^DATA_NAMESPACE=from-dotenv$/d' "$OVR/ctrl/.env"
|
||||
check "the caller beats rig.env" "from-caller" \
|
||||
"$(ov_key ADDONS OVERLAY=local/My_Env ADDONS=from-caller)"
|
||||
|
||||
# Identity follows the overlay's folder, so one rig serves several without collisions.
|
||||
check "identity follows the overlay's folder" "my-env kind-my-env" \
|
||||
"$(in_ov env OVERLAY=local/My_Env bash ports.sh active | awk '{print $1, $2}')"
|
||||
check "paths in rig.env are relative to the overlay" "$OVR/local/My_Env/k8s/prod" \
|
||||
"$(in_ov env OVERLAY=local/My_Env bash ports.sh active | awk '{print $7}')"
|
||||
check "a named overlay that does not exist is an error" "yes" \
|
||||
"$(in_ov env OVERLAY=local/nope bash ports.sh active >/dev/null 2>&1 && echo no || echo yes)"
|
||||
printf 'PROFILE=x\n' > "$OV/bad-rig.env"; mkdir -p "$OVR/local/bad"; cp "$OV/bad-rig.env" "$OVR/local/bad/rig.env"
|
||||
check "rig.env may not choose the profile or the overlay" "yes" \
|
||||
"$(in_ov env OVERLAY=local/bad bash ports.sh active >/dev/null 2>&1 && echo no || echo yes)"
|
||||
|
||||
# Addons: the overlay's is found before rig's own, and runs from rig's ctrl/.
|
||||
check "an overlay's addon comes before rig's of the same name" \
|
||||
"overlay-metallb from $OVR/ctrl with $OVR/ctrl" \
|
||||
"$(in_ov env OVERLAY=local/My_Env ADDONS=metallb bash addons.sh install 2>&1 | grep '^overlay-metallb')"
|
||||
|
||||
# ctrl/.env is this rig's: a pinned block would follow every overlay.
|
||||
check "ports.sh persist refuses while an overlay is set" "yes" \
|
||||
"$(in_ov env OVERLAY=local/My_Env bash ports.sh persist >/dev/null 2>&1 && echo no || echo yes)"
|
||||
|
||||
# make's $(shell) must see an OVERLAY given as a make argument (make < 4.4 does not pass it).
|
||||
check "make -n tilt OVERLAY=... asks for the overlay's context" "kind-data" \
|
||||
"$(cd .. && make --no-print-directory -n tilt OVERLAY=examples/data 2>/dev/null | grep -m1 'tilt ' | sed -n 's/.*--context \([^ ]*\).*/\1/p')"
|
||||
|
||||
# rig reads an overlay and never writes into it; its values never reach a committed kit.
|
||||
sum_ov() { (cd "$OVR/local/My_Env" && find . -type f | sort | xargs sha256sum | sha256sum); }
|
||||
before=$(sum_ov)
|
||||
echo 'OVERLAY=local/My_Env' >> "$OVR/ctrl/.env"
|
||||
in_ov bash ports.sh active >/dev/null 2>&1
|
||||
in_ov bash addons.sh list >/dev/null 2>&1
|
||||
in_ov bash -c 'source ./lib/config.sh; load_config >/dev/null; render_kind_config >/dev/null' 2>/dev/null
|
||||
in_ov bash standalone.sh write >/dev/null 2>&1
|
||||
in_ov bash standalone.sh export "$OV/export" >/dev/null 2>&1
|
||||
check "rig writes nothing into an overlay" "$before" "$(sum_ov)"
|
||||
check "an overlay's values never reach a committed kit" "0" \
|
||||
"$(grep -rlE 'selftest-overlay-sentinel|local/My_Env' "$OVR/standalone" 2>/dev/null | wc -l)"
|
||||
check "a committed kit holds no path of this machine" "0" \
|
||||
"$(grep -rlF "$OVR" "$OVR/standalone" 2>/dev/null | wc -l)"
|
||||
|
||||
|
||||
note "the Tiltfile hardcodes nothing"
|
||||
# The Tiltfile asks ports.sh for its context; a literal kind-<name> would undo that.
|
||||
check "no literal kind-<name>" "0" "$(grep -cE "['\"]kind-[a-z0-9]" Tiltfile)"
|
||||
check "guards on the variable" "1" "$(grep -c 'allow_k8s_contexts(CTX)' Tiltfile)"
|
||||
check "asks ports.sh for facts" "1" "$(grep -c "local('bash ports.sh active'" Tiltfile)"
|
||||
check "hands over to the overlay's Tiltfile" "1" "$(grep -c "include(OVERLAY + '/Tiltfile')" Tiltfile)"
|
||||
|
||||
|
||||
note "standalone kits are generated, current, and call only real verbs"
|
||||
# A kit left stale by a change to rig fails here, not on another machine.
|
||||
check "every kit matches what rig generates now" "yes" \
|
||||
"$(bash standalone.sh check >/dev/null 2>&1 && echo yes || echo "no — run make standalone")"
|
||||
|
||||
# Every kit Makefile target must call a verb its script's own dispatch accepts.
|
||||
verbs_of() {
|
||||
sed -n '/^case "\$cmd" in/,/^esac/p' "$1" | grep -oE '^ [a-z]+\)' | tr -d ' )'
|
||||
}
|
||||
kits=0
|
||||
for mk in ../standalone/*/Makefile; do
|
||||
[ -f "$mk" ] || continue
|
||||
kit=$(dirname "$mk"); kits=$((kits + 1))
|
||||
for target in $(grep -oE '^[a-z][a-z-]*:' "$mk" | tr -d ':' | grep -vx help); do
|
||||
line="$(make --no-print-directory -s -n -f "$mk" "$target" 2>/dev/null | head -1)"
|
||||
script=$(basename "$(printf '%s' "$line" | awk '{print $2}')")
|
||||
verb=$(printf '%s' "$line" | awk '{print $NF}')
|
||||
check "$(basename "$kit"): make $target -> $script $verb, a verb it accepts" "yes" \
|
||||
"$(verbs_of "$kit/$script" | grep -qx "$verb" && echo yes || echo "no: '$verb'")"
|
||||
done
|
||||
check "$(basename "$kit"): no \`mini\` target, which already means minimal footprint" "0" \
|
||||
"$(grep -cE '^mini:' "$mk")"
|
||||
done
|
||||
check "there is a kit for every profile" "$(config_profiles | wc -l)" "$kits"
|
||||
|
||||
# An export carries this machine's choices but never its credentials; committed kits carry neither.
|
||||
# Proven with sentinel values in a scratch copy, since the real ctrl/.env may leave them empty.
|
||||
SX="$TMP/export-proof"; copy_rig "$SX/rig"
|
||||
mkdir -p "$SX/selftest-sentinel-choice/overlays/dev" # a named MANIFESTS_DIR must exist
|
||||
cat >> "$SX/rig/ctrl/.env" <<'EOF'
|
||||
REGISTRY_USER=selftest-sentinel-user
|
||||
REGISTRY_PASSWORD=selftest-sentinel-password
|
||||
MANIFESTS_DIR=../selftest-sentinel-choice/overlays/dev
|
||||
EOF
|
||||
( cd "$SX/rig/ctrl" && bash standalone.sh export "$SX/out" >/dev/null 2>&1 )
|
||||
count_in() { grep -rcF -- "$1" "$2" 2>/dev/null | awk -F: '{s+=$2} END{print s+0}'; }
|
||||
check "export: carries this machine's choices" "yes" \
|
||||
"$([ "$(count_in selftest-sentinel-choice "$SX/out")" -gt 0 ] && echo yes || echo no)"
|
||||
check "export: carries no credential" "0" \
|
||||
"$(( $(count_in selftest-sentinel-user "$SX/out") + $(count_in selftest-sentinel-password "$SX/out") ))"
|
||||
check "per-profile kits: carry neither, whatever this machine has" "0" \
|
||||
"$( (cd "$SX/rig/ctrl" && source ./lib/config.sh && for p in $(config_profiles); do config_snapshot "$p"; done) \
|
||||
| grep -cE 'selftest-sentinel-(choice|user|password)')"
|
||||
check "export: refuses to write inside the repository" "yes" \
|
||||
"$( (bash standalone.sh export ../standalone/selftest-mine >/dev/null 2>&1) && echo no || echo yes)"
|
||||
|
||||
|
||||
note "rig's addons apply verified files, never URLs"
|
||||
# The offline profile must need no network for manifests: each addon asks deps.sh for a
|
||||
# pinned manifest, verified on disk (versions.md). Their images still need preloading.
|
||||
check "no rig addon applies a URL" "0" \
|
||||
"$(cat addons/*.sh | grep -cE 'apply -f "?https?://')"
|
||||
check "every manifest an addon asks for is pinned with a sum" "" \
|
||||
"$(for n in $(grep -ohE 'deps\.sh manifest [A-Z_]+' addons/*.sh | awk '{print $3}' | sort -u); do
|
||||
grep -q "^${n}_MANIFEST_URL=" versions.env && grep -q "^${n}_MANIFEST_SHA256=[0-9a-f]\{64\}$" versions.env \
|
||||
|| printf '%s ' "$n"; done)"
|
||||
|
||||
|
||||
note "withdrawn stays withdrawn (STALE.md)"
|
||||
# One check per entry; the reasoning is in STALE.md, not here.
|
||||
check "✖ S1 rig's Tiltfile has no Images section of its own" "0" "$(grep -c '^# ── Images' Tiltfile)"
|
||||
check "✖ S2 local/ is where overlays live, and ignored" "yes" \
|
||||
"$(grep -qx '/local/' ../.gitignore && echo yes || echo no)"
|
||||
# Patterns assembled from fragments so this file does not match itself.
|
||||
COPIES_PAT="$(printf '%s' 'ac' 'me-rig|ac' 'mebank')"
|
||||
HOUSE_PAT="$(printf '%s' 'semes' 'ter|local' '\.ar\b')"
|
||||
check "✖ S2 no example environment name from the copies era" "0" \
|
||||
"$(cd .. && grep -rIlE "$COPIES_PAT" . --exclude-dir=def --exclude-dir=local --exclude=STALE.md 2>/dev/null | wc -l)"
|
||||
check "✖ S3 ctrl/addons makes the cluster work, nothing more" "cert-manager metallb metrics-server" \
|
||||
"$(ls addons/ | sed 's/\.sh$//' | sort | xargs)"
|
||||
check "✖ S3 versions.env pins no workload image" "0" \
|
||||
"$(grep -cE '^(POSTGRES|REDIS|AIRFLOW)_IMAGE=' versions.env)"
|
||||
check "✖ S4 no namespace named after the cluster" "0" "$(grep -c "CLUSTER + ':namespace'" Tiltfile)"
|
||||
check "✖ S5 rig's examples left ctrl/k8s" "no" "$([ -d k8s/overlays ] && echo yes || echo no)"
|
||||
check "✖ S5 .env.example does not pin MANIFESTS_DIR" "0" "$(grep -c '^MANIFESTS_DIR=' .env.example)"
|
||||
check "✖ S6 no client or data example profile" "0" \
|
||||
"$(ls env.d/ | grep -cE '^(client|data)\.')"
|
||||
check "✖ S7 no house path or host name in rig" "0" \
|
||||
"$(cd .. && grep -rIlE "$HOUSE_PAT" . --exclude-dir=def --exclude-dir=local --exclude=STALE.md 2>/dev/null | wc -l)"
|
||||
|
||||
|
||||
note "the dev loop parses — needs tilt and kubectl, not a cluster"
|
||||
# A throwaway kubeconfig with kind-named entries (Tilt trusts kind contexts) and a
|
||||
# kubectl that swallows `apply`: the Tiltfile evaluates for real, nothing is contacted.
|
||||
# The second run is a copy under another name, the case that once failed at load.
|
||||
if ! command -v tilt >/dev/null || ! command -v kubectl >/dev/null; then
|
||||
printf ' skip tilt or kubectl is not installed\n'
|
||||
else
|
||||
FK="$TMP/fake-kube"; mkdir -p "$FK"
|
||||
real_kubectl=$(command -v kubectl)
|
||||
printf '#!/usr/bin/env bash\nfor a in "$@"; do [ "$a" = apply ] && { cat >/dev/null; exit 0; }; done\nexec %q "$@"\n' \
|
||||
"$real_kubectl" > "$FK/kubectl"
|
||||
chmod +x "$FK/kubectl"
|
||||
parses() { # cluster-name [env...] -> the manifests Tilt would deploy, or the error
|
||||
local name="$1"; shift
|
||||
cat > "$FK/kubeconfig" <<EOF
|
||||
apiVersion: v1
|
||||
kind: Config
|
||||
clusters: [{name: kind-$name, cluster: {server: "https://127.0.0.1:9"}}]
|
||||
contexts: [{name: kind-$name, context: {cluster: kind-$name, user: kind-$name}}]
|
||||
users: [{name: kind-$name, user: {token: selftest}}]
|
||||
current-context: kind-$name
|
||||
EOF
|
||||
env "$@" KUBECONFIG="$FK/kubeconfig" PATH="$FK:$PATH" \
|
||||
timeout 120 tilt alpha tiltfile-result --context "kind-$name" > "$FK/out.json" 2> "$FK/err" \
|
||||
&& grep -o '"Name": *"[^"]*"' "$FK/out.json" | sed 's/.*"\([^"]*\)"$/\1/' | sort -u | xargs \
|
||||
|| grep -m1 -iE 'error|no object' "$FK/err"
|
||||
}
|
||||
check "the starter overlay parses" "example-service infra uncategorized" "$(parses rig)"
|
||||
check "and under another name" "example-service infra uncategorized" \
|
||||
"$(parses selftest-copy CLUSTER=selftest-copy)"
|
||||
check "the data overlay parses" "items-api uncategorized" "$(parses data OVERLAY=examples/data)"
|
||||
fi
|
||||
|
||||
|
||||
note "the examples are overlays that work as shipped"
|
||||
# They are what a real overlay is copied from, so they must at least parse.
|
||||
bad=""
|
||||
for f in ../examples/*/addons/*.sh; do [ -e "$f" ] && { bash -n "$f" 2>/dev/null || bad+="$f "; }; done
|
||||
check "every example addon parses" "" "$bad"
|
||||
if command -v python3 >/dev/null; then
|
||||
bad=""
|
||||
for f in ../examples/*/dags/*.py; do
|
||||
[ -e "$f" ] && { python3 -c 'import ast, sys; ast.parse(open(sys.argv[1]).read())' "$f" 2>/dev/null || bad+="$f "; }
|
||||
done
|
||||
check "every example DAG parses" "" "$bad"
|
||||
else
|
||||
printf ' skip python3 is not installed\n'
|
||||
fi
|
||||
|
||||
|
||||
printf '\n'
|
||||
if [ "$rc" -eq 0 ]; then
|
||||
printf '%d checks passed — rig still does what it says\n' "$passed"
|
||||
else
|
||||
printf 'FAILED — a decision above has drifted; read the comment next to it\n' >&2
|
||||
fi
|
||||
exit "$rc"
|
||||
@@ -1,249 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Prepare a machine to run rig, and say plainly what worked, what was already
|
||||
# done, and what is left for a human.
|
||||
#
|
||||
# This is the grouped entry point: `make setup`. Every step is idempotent and
|
||||
# independently checked, so running it twice is safe and running it on a
|
||||
# half-configured machine finishes the job rather than starting over.
|
||||
#
|
||||
# It deliberately does NOT abort on the first failure. A setup script that dies
|
||||
# at step 2 hides the fact that steps 4 and 5 were also going to fail — and on
|
||||
# an unfamiliar machine, the full picture is the whole point. Failures are
|
||||
# collected and reported together, and the exit code reflects the worst outcome.
|
||||
#
|
||||
# The same script runs inside a fresh throwaway distro (newbox), so the
|
||||
# provisioning path and the everyday path cannot drift apart.
|
||||
#
|
||||
# Usage:
|
||||
# setup.sh # host checks + the dev toolchain
|
||||
# setup.sh core # kubectl and jq only — no cluster tooling
|
||||
# setup.sh --share-docker # ...and offer this distro's Docker to others
|
||||
# setup.sh --cluster # ...and bring the cluster up
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")"
|
||||
|
||||
source ./lib/config.sh
|
||||
load_config
|
||||
|
||||
WITH_SHARE=0
|
||||
WITH_CLUSTER=0
|
||||
# Cluster tooling is not wanted everywhere: a managed or corporate-issued
|
||||
# machine may legitimately want kubectl and nothing that builds clusters.
|
||||
TIER=dev
|
||||
for a in "$@"; do
|
||||
case "$a" in
|
||||
core|dev) TIER="$a" ;;
|
||||
--share-docker) WITH_SHARE=1 ;;
|
||||
--cluster) WITH_CLUSTER=1 ;;
|
||||
*) echo "unknown option: $a" >&2; exit 1 ;;
|
||||
esac
|
||||
done
|
||||
|
||||
if [ "$TIER" = "core" ] && [ "$WITH_CLUSTER" -eq 1 ]; then
|
||||
echo "core tier installs no cluster tooling, so --cluster cannot work" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# ── step framework ─────────────────────────────────────────────────────────
|
||||
# Statuses are deliberately distinct: "already" and "done" both mean success but
|
||||
# tell you very different things about the machine you are on.
|
||||
STEP_NAMES=()
|
||||
STEP_STATUS=()
|
||||
STEP_NOTE=()
|
||||
WORST=0
|
||||
|
||||
record() {
|
||||
STEP_NAMES+=("$1"); STEP_STATUS+=("$2"); STEP_NOTE+=("${3:-}")
|
||||
# Only a genuine failure is a non-zero exit. "manual" means the machine is
|
||||
# fine and you have something to do — reporting that as an error makes the
|
||||
# whole run look broken and trains people to ignore the output.
|
||||
[ "$2" = "fail" ] && WORST=1 || true
|
||||
local mark
|
||||
case "$2" in
|
||||
already) mark=" ok " ;;
|
||||
done) mark=" done " ;;
|
||||
skip) mark=" skip " ;;
|
||||
manual) mark="MANUAL" ;;
|
||||
fail) mark=" FAIL " ;;
|
||||
esac
|
||||
printf "[%s] %-22s %s\n" "$mark" "$1" "${3:-}"
|
||||
}
|
||||
|
||||
# ── steps ──────────────────────────────────────────────────────────────────
|
||||
|
||||
step_host() {
|
||||
local out
|
||||
if ! out=$(bash ./deps.sh detect 2>&1); then
|
||||
record host fail "detection failed"
|
||||
return
|
||||
fi
|
||||
# Anything flagged with '!' needs a human; surface the count here
|
||||
# and the detail below rather than burying it.
|
||||
local warns; warns=$(echo "$out" | grep -c '^\s*!' || true)
|
||||
HOST_DETAIL="$out"
|
||||
if [ "$warns" -gt 0 ]; then
|
||||
record host manual "$warns item(s) need attention — see below"
|
||||
else
|
||||
record host already "no problems detected"
|
||||
fi
|
||||
}
|
||||
|
||||
step_toolchain() {
|
||||
local want="kubectl jq"
|
||||
[ "$TIER" = "dev" ] && want="$want kind tilt"
|
||||
|
||||
local missing=""
|
||||
for b in $want; do
|
||||
command -v "$b" >/dev/null 2>&1 || missing="$missing $b"
|
||||
done
|
||||
|
||||
if [ -z "$missing" ]; then
|
||||
record toolchain already "$TIER: $want"
|
||||
return
|
||||
fi
|
||||
|
||||
if bash ./deps.sh install "$TIER" >/tmp/rig-deps.$$ 2>&1; then
|
||||
local still=""
|
||||
for b in $want; do
|
||||
[ -x "${OUT_BIN:-$HOME/.local/bin}/$b" ] || still="$still $b"
|
||||
done
|
||||
if [ -n "$still" ]; then
|
||||
record toolchain fail "still missing:$still (see /tmp/rig-deps.$$)"
|
||||
else
|
||||
record toolchain done "$TIER, installed:$missing"
|
||||
rm -f "/tmp/rig-deps.$$"
|
||||
fi
|
||||
else
|
||||
record toolchain fail "install failed — see /tmp/rig-deps.$$"
|
||||
fi
|
||||
}
|
||||
|
||||
step_path() {
|
||||
local bin="${OUT_BIN:-$HOME/.local/bin}"
|
||||
case ":$PATH:" in
|
||||
*":$bin:"*) ;;
|
||||
*) record path manual "add to ~/.bashrc: export PATH=\"$bin:\$PATH\""; return ;;
|
||||
esac
|
||||
if grep -qs "$bin" "$HOME/.bashrc" "$HOME/.profile" 2>/dev/null; then
|
||||
record path already "$bin on PATH and persisted"
|
||||
else
|
||||
record path manual "on PATH now, but not persisted in ~/.bashrc"
|
||||
fi
|
||||
}
|
||||
|
||||
step_docker() {
|
||||
if ! command -v docker >/dev/null 2>&1; then
|
||||
record docker fail "no docker cli — this is the one prerequisite rig cannot install"
|
||||
return
|
||||
fi
|
||||
if docker info >/dev/null 2>&1; then
|
||||
record docker already "$(docker version --format '{{.Server.Version}}' 2>/dev/null)"
|
||||
else
|
||||
record docker fail "daemon unreachable (in the docker group? logged out and back in?)"
|
||||
fi
|
||||
}
|
||||
|
||||
step_share_docker() {
|
||||
if [ "$WITH_SHARE" -ne 1 ]; then
|
||||
record docker-share skip "not requested (--share-docker)"
|
||||
return
|
||||
fi
|
||||
if ! grep -qi microsoft /proc/version 2>/dev/null; then
|
||||
record docker-share skip "not WSL — sharing only applies between WSL distros"
|
||||
return
|
||||
fi
|
||||
if [ -f /etc/systemd/system/docker.service.d/10-rig-shared-socket.conf ]; then
|
||||
record docker-share already "this distro is offering its Docker to others"
|
||||
return
|
||||
fi
|
||||
# Needs root, and asking mid-script is worse than telling the user the
|
||||
# single command to run.
|
||||
if [ "$(id -u)" -ne 0 ] && ! sudo -n true 2>/dev/null; then
|
||||
record docker-share manual "run: sudo bash ctrl/dockerhost.sh share"
|
||||
return
|
||||
fi
|
||||
if sudo bash ./dockerhost.sh share >/tmp/rig-share.$$ 2>&1; then
|
||||
record docker-share done "this distro now owns the shared Docker"
|
||||
rm -f "/tmp/rig-share.$$"
|
||||
else
|
||||
record docker-share fail "see /tmp/rig-share.$$"
|
||||
fi
|
||||
}
|
||||
|
||||
step_ports() {
|
||||
local busy=""
|
||||
for entry in "HTTP:$HTTP_PORT" "HTTPS:$HTTPS_PORT" "TILT:$TILT_PORT" "REGISTRY:$REGISTRY_PORT"; do
|
||||
local p="${entry#*:}"
|
||||
if command -v ss >/dev/null 2>&1 && ss -ltn "sport = :$p" 2>/dev/null | grep -q LISTEN; then
|
||||
busy="$busy ${entry%%:*}($p)"
|
||||
fi
|
||||
done
|
||||
if [ -n "$busy" ]; then
|
||||
record ports fail "in use:$busy — override in ctrl/.env or rename the directory"
|
||||
else
|
||||
record ports already "$HTTP_PORT-$REGISTRY_PORT free"
|
||||
fi
|
||||
}
|
||||
|
||||
step_cluster() {
|
||||
if [ "$TIER" = "core" ]; then
|
||||
record cluster skip "core tier — no cluster tooling on this machine"
|
||||
return
|
||||
fi
|
||||
if [ "$WITH_CLUSTER" -ne 1 ]; then
|
||||
record cluster skip "not requested (--cluster)"
|
||||
return
|
||||
fi
|
||||
if kind get clusters 2>/dev/null | grep -qx "$CLUSTER"; then
|
||||
record cluster already "'$CLUSTER' exists"
|
||||
return
|
||||
fi
|
||||
if bash ./cluster.sh up >/tmp/rig-cluster.$$ 2>&1; then
|
||||
record cluster done "'$CLUSTER' created"
|
||||
rm -f "/tmp/rig-cluster.$$"
|
||||
else
|
||||
record cluster fail "see /tmp/rig-cluster.$$"
|
||||
fi
|
||||
}
|
||||
|
||||
# ── run ────────────────────────────────────────────────────────────────────
|
||||
|
||||
echo "setting up '$CLUSTER'"
|
||||
echo
|
||||
HOST_DETAIL=""
|
||||
step_host
|
||||
step_toolchain
|
||||
step_path
|
||||
step_docker
|
||||
step_share_docker
|
||||
step_ports
|
||||
step_cluster
|
||||
|
||||
echo
|
||||
if [ -n "$HOST_DETAIL" ]; then
|
||||
echo "host detail"
|
||||
echo "$HOST_DETAIL" | sed 's/^/ /'
|
||||
echo
|
||||
fi
|
||||
|
||||
# Repeat only what still needs action, so the tail of the output is a to-do list
|
||||
# rather than a transcript.
|
||||
outstanding=0
|
||||
for i in "${!STEP_NAMES[@]}"; do
|
||||
case "${STEP_STATUS[$i]}" in
|
||||
fail|manual)
|
||||
[ "$outstanding" -eq 0 ] && echo "outstanding:"
|
||||
outstanding=1
|
||||
printf " %-8s %-16s %s\n" "${STEP_STATUS[$i]}" "${STEP_NAMES[$i]}" "${STEP_NOTE[$i]}"
|
||||
;;
|
||||
esac
|
||||
done
|
||||
|
||||
if [ "$outstanding" -eq 0 ]; then
|
||||
echo "ready. next: make cluster up && make docs"
|
||||
else
|
||||
echo
|
||||
echo "(nothing was aborted — every step ran so the list above is complete)"
|
||||
fi
|
||||
|
||||
exit "$WORST"
|
||||
350
rig/ctrl/standalone.sh
Normal file
350
rig/ctrl/standalone.sh
Normal file
@@ -0,0 +1,350 @@
|
||||
#!/usr/bin/env bash
|
||||
# Generate standalone kits: rig's tools flattened into single files, one folder per profile.
|
||||
# Usage: standalone.sh write|check generate (or diff) standalone/<profile>/
|
||||
# standalone.sh export DIR one kit for this machine's config, no credentials, outside the repo
|
||||
# Notes: docs/notes/standalone.md
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")"
|
||||
|
||||
CTRL="$PWD"
|
||||
ROOT="$(cd .. && pwd)"
|
||||
OUT="$ROOT/standalone"
|
||||
SELF_REL="ctrl/${0##*/}"
|
||||
|
||||
GENERATED_TAG="GENERATED by make standalone — do not edit"
|
||||
|
||||
# The contract's own functions: questions rig answers FOR this generator. They
|
||||
# are never carried into a kit — load_config is replaced by the frozen one, and
|
||||
# the rest mean nothing without rig's tree. The only names this file knows.
|
||||
CONTRACT_FUNCS="load_config config_profiles config_snapshot config_freeze config_current_profile config_left_out"
|
||||
FROZEN_OPEN="# ── configuration, frozen"
|
||||
FROZEN_CLOSE="# ── end of frozen configuration"
|
||||
|
||||
refuse() { echo >&2; echo "standalone: refusing — $*" >&2; exit 1; }
|
||||
|
||||
# A clean bash with nothing from the caller's shell in it. What the kit carries
|
||||
# must not depend on who ran the generator or what they had exported.
|
||||
clean_bash() { env -i PATH="$PATH" HOME="$HOME" CONTRACT_FUNCS="$CONTRACT_FUNCS" bash --noprofile --norc "$@"; }
|
||||
|
||||
# ── 1. entry points ────────────────────────────────────────────────────────
|
||||
entries() {
|
||||
grep -rlE --include='*.sh' '^# rig:standalone [a-z0-9-]+ [a-z0-9-]+' . 2>/dev/null \
|
||||
| sed 's|^\./||' | LC_ALL=C sort
|
||||
}
|
||||
marker_of() { # entry -> "kit verb"
|
||||
sed -nE 's/^# rig:standalone ([a-z0-9-]+) ([a-z0-9-]+).*/\1 \2/p' "$1" | head -1
|
||||
}
|
||||
|
||||
# ── 2. the libraries an entry point sources ────────────────────────────────
|
||||
# Only the entry point's own `source` lines are read as text. Everything those
|
||||
# libraries pull in is resolved by bash when they are sourced in step 3.
|
||||
libs_of() { # entry -> one resolved lib path per line, relative to ctrl/
|
||||
local entry="$1" dir line n path
|
||||
dir=$(dirname "$entry")
|
||||
while IFS=: read -r n line; do
|
||||
path=$(printf '%s' "$line" | sed -E 's/^[[:space:]]*(source|\.)[[:space:]]+//; s/[[:space:]]+(#.*)?$//')
|
||||
path=${path#\"}; path=${path%\"}; path=${path#\'}; path=${path%\'}
|
||||
case "$path" in
|
||||
*'$'*) refuse "$entry:$n sources '$path' — a path with a variable in it cannot be resolved; name the file" ;;
|
||||
esac
|
||||
case "$path" in
|
||||
*.sh) ;;
|
||||
*) refuse "$entry:$n sources '$path' directly — only libraries (.sh) may be sourced; configuration has to enter through load_config" ;;
|
||||
esac
|
||||
path="$dir/${path#./}"; path=${path#./}
|
||||
[ -f "$path" ] || refuse "$entry:$n sources '$path', which does not exist"
|
||||
printf '%s\n' "$path"
|
||||
done < <(grep -nE '^[[:space:]]*(source|\.)[[:space:]]+[^=]' "$entry" || true)
|
||||
}
|
||||
|
||||
# Into the global array `libs`. Not `mapfile < <(libs_of ...)`: a refusal inside
|
||||
# a process substitution only ends that subshell, so generation would carry on
|
||||
# past it and fail later with a message about something else entirely.
|
||||
libs_into() {
|
||||
local out
|
||||
out=$(libs_of "$1") || exit 1
|
||||
libs=()
|
||||
[ -n "$out" ] && mapfile -t libs <<< "$out"
|
||||
return 0
|
||||
}
|
||||
|
||||
# ── 3. what the libraries define, read back from bash itself ───────────────
|
||||
# The frozen config replaces load_config, and the generator's own two questions
|
||||
# are useless inside a kit, so none of the three is carried.
|
||||
lib_defs() { # entry lib... -> declare -p globals, then declare -f functions
|
||||
local entry="$1"; shift
|
||||
( cd "$(dirname "$entry")" && clean_bash -c '
|
||||
skip_var() { case "$1" in CONTRACT_FUNCS|BASH*|FUNCNAME|PIPESTATUS|LINENO|RANDOM|SRANDOM|SECONDS|EPOCH*|HISTCMD|COLUMNS|LINES|PWD|OLDPWD|_|SHLVL|OPTIND|OPTERR|IFS|PS4|PATH|HOME|v|f|l|before_v|before_f) return 0 ;; esac; return 1; }
|
||||
before_v=" $(compgen -v | tr "\n" " ") "
|
||||
before_f=" $(compgen -A function | tr "\n" " ") "
|
||||
for l in "$@"; do source "$l" || { echo "__FAIL__ sourcing $l" ; exit 1; }; done
|
||||
for v in $(compgen -v); do
|
||||
skip_var "$v" && continue
|
||||
case "$before_v" in *" $v "*) continue ;; esac
|
||||
declare -p "$v"
|
||||
done
|
||||
for f in $(compgen -A function); do
|
||||
case "$before_f" in *" $f "*) continue ;; esac
|
||||
case " skip_var $CONTRACT_FUNCS " in *" $f "*) continue ;; esac
|
||||
declare -f "$f"
|
||||
done
|
||||
' _ "$@" ) || refuse "$entry: its libraries could not be sourced cleanly"
|
||||
}
|
||||
|
||||
# ── 4. ask rig for profiles and resolved config ────────────────────────────
|
||||
ask() { # entry lib... -- function args... -> that function's stdout
|
||||
local entry="$1"; shift
|
||||
local libs=() a
|
||||
while [ $# -gt 0 ] && [ "$1" != -- ]; do libs+=("$1"); shift; done
|
||||
shift
|
||||
( cd "$(dirname "$entry")" && clean_bash -c '
|
||||
n=0; for a in "$@"; do n=$((n+1)); [ "$a" = -- ] && break; done
|
||||
for l in "${@:1:$((n-1))}"; do source "$l"; done
|
||||
shift "$n"
|
||||
declare -F "$1" >/dev/null || exit 3
|
||||
"$@"
|
||||
' _ "${libs[@]}" -- "$@" )
|
||||
}
|
||||
|
||||
# ── 5. assemble one kit file ───────────────────────────────────────────────
|
||||
assemble() { # entry profile out-file lib...
|
||||
local entry="$1" profile="$2" dest="$3"; shift 3
|
||||
local libs=("$@") calls_config=no
|
||||
grep -qE '(^|[^A-Za-z0-9_])load_config([^A-Za-z0-9_]|$)' "$entry" && calls_config=yes
|
||||
|
||||
{
|
||||
echo '#!/usr/bin/env bash'
|
||||
echo "# $GENERATED_TAG"
|
||||
echo "#"
|
||||
echo "# $(basename "$dest") for ${KIT_LABEL:-profile '$profile'}, flattened from:"
|
||||
echo "# ctrl/$entry"
|
||||
local l; for l in ${libs[@]+"${libs[@]}"}; do echo "# ctrl/$l"; done
|
||||
echo "# Edit those and run \`make standalone\`. Changes made here are lost, and"
|
||||
echo "# \`make selftest\` fails while this file differs from what rig generates."
|
||||
echo
|
||||
|
||||
if [ ${#libs[@]} -gt 0 ]; then
|
||||
echo "# ── from the libraries ──"
|
||||
lib_defs "$entry" "${libs[@]}"
|
||||
echo
|
||||
fi
|
||||
|
||||
if [ "$calls_config" = yes ]; then
|
||||
local frozen
|
||||
frozen=$(ask "$entry" ${libs[@]+"${libs[@]}"} -- config_freeze "${FREEZE_ARG:-$profile}") \
|
||||
|| refuse "ctrl/$entry calls load_config, but its libraries do not answer config_freeze ${FREEZE_ARG:-$profile}"
|
||||
echo "$FROZEN_OPEN for ${KIT_LABEL:-profile '$profile'} ──"
|
||||
printf '%s\n' "$frozen"
|
||||
echo "$FROZEN_CLOSE ──"
|
||||
echo
|
||||
fi
|
||||
|
||||
echo "# ── ctrl/$entry ──"
|
||||
# The entry point itself, minus its shebang and marker, with each source
|
||||
# line it made replaced by a note — what it sourced is already above.
|
||||
awk '
|
||||
NR == 1 && /^#!/ { next }
|
||||
/^# rig:standalone / { next }
|
||||
/^[[:space:]]*(source|\.)[[:space:]]+[^=]/ { print "# (sourced library inlined above)"; next }
|
||||
{ print }
|
||||
' "$entry"
|
||||
} > "$dest"
|
||||
chmod +x "$dest"
|
||||
}
|
||||
|
||||
# ── 6. the kit's Makefile, from the markers ────────────────────────────────
|
||||
verbs_of() { # entry -> its top-level dispatch arms
|
||||
awk '/^case / { inb=1; next } /^esac/ { inb=0 } inb && match($0, /^ [a-z][a-z-]*\)/) { v=substr($0, 5, RLENGTH-5); printf "%s%s", (n++ ? "|" : ""), v }' "$1"
|
||||
}
|
||||
makefile() { # out-dir entry...
|
||||
local dir="$1"; shift
|
||||
local e kit verb target verbs targets=""
|
||||
for e in "$@"; do targets+=" $(basename "$e" .sh)"; done
|
||||
{
|
||||
echo "# $GENERATED_TAG"
|
||||
echo "#"
|
||||
echo "# Shorthand for the scripts beside it; they run without it. Every target"
|
||||
echo "# calls a verb its script accepts — read from that script's own dispatch."
|
||||
echo
|
||||
echo 'HERE := $(dir $(abspath $(lastword $(MAKEFILE_LIST))))'
|
||||
echo 'ARGS := $(wordlist 2,$(words $(MAKECMDGOALS)),$(MAKECMDGOALS))'
|
||||
echo 'ifneq ($(ARGS),)'
|
||||
echo '$(eval $(ARGS):;@:)'
|
||||
echo '.PHONY: $(ARGS)'
|
||||
echo 'endif'
|
||||
echo
|
||||
echo '.DEFAULT_GOAL := help'
|
||||
echo ".PHONY: help$targets"
|
||||
echo
|
||||
echo 'help: ## list targets'
|
||||
printf '\t%s\n' "@grep -hE '^[a-z][a-z-]*:.*?##' \$(MAKEFILE_LIST) | sed 's/:.*##/\\t/' | expand -t16"
|
||||
for e in "$@"; do
|
||||
read -r kit verb <<< "$(marker_of "$e")"
|
||||
target=$(basename "$e" .sh)
|
||||
verbs=$(verbs_of "$e")
|
||||
echo
|
||||
printf '%-30s ## %s.sh [%s] (default %s)\n' "$target:" "$kit" "${verbs:-?}" "$verb"
|
||||
printf '\tbash $(HERE)%s.sh $(or $(ARGS),%s)\n' "$kit" "$verb"
|
||||
done
|
||||
} > "$dir/Makefile"
|
||||
}
|
||||
|
||||
# ── 7. prove a kit stands alone ────────────────────────────────────────────
|
||||
verify_kit() { # dir profile entry...
|
||||
local dir="$1" profile="$2"; shift 2
|
||||
local e kit verb f bad smoke rc
|
||||
for e in "$@"; do
|
||||
read -r kit verb <<< "$(marker_of "$e")"
|
||||
f="$dir/$kit.sh"
|
||||
|
||||
bash -n "$f" 2>/dev/null || refuse "$profile/$kit.sh does not parse: $(bash -n "$f" 2>&1 | head -1)"
|
||||
|
||||
# Code only: comments are free to mention anything, and the frozen block
|
||||
# is data — a value that happens to hold a path is harmless unless code
|
||||
# opens it, and opening it is what the smoke run below would catch.
|
||||
bad=$(awk -v fz_open="$FROZEN_OPEN" -v fz_close="$FROZEN_CLOSE" '
|
||||
index($0, fz_open) == 1 { fz=1; next }
|
||||
index($0, fz_close) == 1 { fz=0; next }
|
||||
fz || /^[[:space:]]*#/ { next }
|
||||
/^[[:space:]]*(source|\.)[[:space:]]+[^=]/ { printf "%d: still sources: %s\n", NR, $0; next }
|
||||
# Rig-relative only. The preceding character may not be "/", so an
|
||||
# absolute system path such as /var/lib/docker is not mistaken for
|
||||
# rig lib/; an explicit ./ or ../ prefix is matched on its own.
|
||||
/(^|[^A-Za-z0-9_.\/])(ctrl\/|lib\/|env\.d\/)|\.\.?\/(ctrl\/|lib\/|env\.d\/)|versions\.env|(^|[^A-Za-z0-9_])\.env([^A-Za-z0-9_]|$)/ {
|
||||
printf "%d: refers into rig'"'"'s tree: %s\n", NR, $0
|
||||
}' "$f" | head -3)
|
||||
[ -z "$bad" ] || refuse "$profile/$kit.sh does not stand alone —"$'\n'"$(printf '%s\n' "$bad" | sed 's/^/ line /')"
|
||||
done
|
||||
|
||||
# The real test: a folder holding only this kit, and nothing else from rig.
|
||||
smoke=$(mktemp -d)
|
||||
cp "$dir"/* "$smoke"/
|
||||
for e in "$@"; do
|
||||
read -r kit verb <<< "$(marker_of "$e")"
|
||||
rc=0
|
||||
out=$( (cd "$smoke" && timeout 120 bash "./$kit.sh" "$verb") 2>&1 ) || rc=$?
|
||||
if [ "$rc" -ne 0 ]; then
|
||||
rm -rf "$smoke"
|
||||
refuse "$profile/$kit.sh $verb exits $rc in an empty directory:"$'\n'"$(printf '%s\n' "$out" | tail -5 | sed 's/^/ /')"
|
||||
fi
|
||||
done
|
||||
( cd "$smoke" && make -s help >/dev/null ) || { rm -rf "$smoke"; refuse "$profile/Makefile: make help fails"; }
|
||||
rm -rf "$smoke"
|
||||
}
|
||||
|
||||
# ── generate ───────────────────────────────────────────────────────────────
|
||||
generate() { # into-dir
|
||||
local into="$1" e profiles="" p kit verb libs
|
||||
local -a all_entries=()
|
||||
while IFS= read -r e; do all_entries+=("$e"); done < <(entries)
|
||||
[ ${#all_entries[@]} -gt 0 ] || refuse "no script under ctrl/ carries a '# rig:standalone <kit> <verb>' marker"
|
||||
|
||||
# Profiles come from whichever entry point's libraries can answer for them.
|
||||
for e in "${all_entries[@]}"; do
|
||||
libs_into "$e"
|
||||
profiles=$(ask "$e" ${libs[@]+"${libs[@]}"} -- config_profiles 2>/dev/null) && [ -n "$profiles" ] && break
|
||||
profiles=""
|
||||
done
|
||||
[ -n "$profiles" ] || refuse "no entry point's libraries answer config_profiles, so there is nothing to generate a kit per"
|
||||
|
||||
for p in $profiles; do
|
||||
mkdir -p "$into/$p"
|
||||
for e in "${all_entries[@]}"; do
|
||||
read -r kit verb <<< "$(marker_of "$e")"
|
||||
libs_into "$e"
|
||||
assemble "$e" "$p" "$into/$p/$kit.sh" ${libs[@]+"${libs[@]}"}
|
||||
done
|
||||
makefile "$into/$p" "${all_entries[@]}"
|
||||
verify_kit "$into/$p" "$p" "${all_entries[@]}"
|
||||
echo " $p: $(cd "$into/$p" && ls | tr '\n' ' ')"
|
||||
done
|
||||
}
|
||||
|
||||
# A kit folder is ours if its Makefile says so. Anything else under standalone/
|
||||
# is left alone, so a hand-written file there is never swept away.
|
||||
is_generated_dir() { grep -qF "$GENERATED_TAG" "$1/Makefile" 2>/dev/null; }
|
||||
|
||||
cmd="${1:-write}"
|
||||
[ $# -gt 0 ] && shift
|
||||
|
||||
case "$cmd" in
|
||||
write)
|
||||
tmp=$(mktemp -d); trap 'rm -rf "$tmp"' EXIT
|
||||
echo "generating kits from rig's current tree"
|
||||
generate "$tmp"
|
||||
mkdir -p "$OUT"
|
||||
for d in "$OUT"/*/; do
|
||||
d=${d%/}; [ -d "$d" ] || continue
|
||||
if is_generated_dir "$d" && [ ! -d "$tmp/${d##*/}" ]; then
|
||||
echo " removed $(basename "$d") — no such profile any more"
|
||||
rm -rf "$d"
|
||||
fi
|
||||
done
|
||||
for d in "$tmp"/*/; do
|
||||
d=${d%/}
|
||||
rm -rf "$OUT/${d##*/}"
|
||||
cp -r "$d" "$OUT/${d##*/}"
|
||||
done
|
||||
echo "wrote standalone/<profile>/ — every kit verified to stand alone"
|
||||
;;
|
||||
check)
|
||||
tmp=$(mktemp -d); trap 'rm -rf "$tmp"' EXIT
|
||||
generate "$tmp" >/dev/null
|
||||
stale=0
|
||||
for d in "$tmp"/*/; do
|
||||
d=${d%/}; p=${d##*/}
|
||||
if ! diff -rq "$d" "$OUT/$p" >/dev/null 2>&1; then
|
||||
echo "stale: standalone/$p — $(diff -rq "$d" "$OUT/$p" 2>&1 | head -1)"
|
||||
stale=1
|
||||
fi
|
||||
done
|
||||
for d in "$OUT"/*/; do
|
||||
d=${d%/}; [ -d "$d" ] || continue
|
||||
if is_generated_dir "$d" && [ ! -d "$tmp/${d##*/}" ]; then
|
||||
echo "stale: standalone/${d##*/} — no such profile any more"; stale=1
|
||||
fi
|
||||
done
|
||||
[ "$stale" -eq 0 ] || { echo "run: make standalone"; exit 1; }
|
||||
echo "every kit is current"
|
||||
;;
|
||||
export)
|
||||
dest="${1:-}"
|
||||
[ -n "$dest" ] || refuse "export needs a directory, outside the repo: make standalone export ~/rig-kit"
|
||||
dest=$(realpath -m "$dest")
|
||||
top=$(git -C "$ROOT" rev-parse --show-toplevel 2>/dev/null || echo "$ROOT")
|
||||
case "$dest/" in
|
||||
"$top"/*) refuse "an export reflects this machine, so it does not go inside the repository — $dest is under $top. The committed per-profile kits are what standalone/ is for." ;;
|
||||
esac
|
||||
if [ -d "$dest" ] && [ -n "$(ls -A "$dest" 2>/dev/null)" ] && ! is_generated_dir "$dest"; then
|
||||
refuse "$dest already holds something that is not a previous export — pick an empty directory"
|
||||
fi
|
||||
|
||||
mapfile -t all_entries < <(entries)
|
||||
[ ${#all_entries[@]} -gt 0 ] || refuse "no script under ctrl/ carries a '# rig:standalone <kit> <verb>' marker"
|
||||
libs_into "${all_entries[0]}"
|
||||
profile=$(ask "${all_entries[0]}" ${libs[@]+"${libs[@]}"} -- config_current_profile) \
|
||||
|| refuse "this machine's configuration does not resolve — run make check"
|
||||
left=$(ask "${all_entries[0]}" ${libs[@]+"${libs[@]}"} -- config_left_out | tr '\n' ' ')
|
||||
|
||||
tmp=$(mktemp -d); trap 'rm -rf "$tmp"' EXIT
|
||||
echo "exporting the configuration this machine runs (profile '$profile')"
|
||||
FREEZE_ARG=--current
|
||||
KIT_LABEL="the configuration exported from $(hostname -s 2>/dev/null || echo this machine) (profile '$profile', local choices included, credentials not)"
|
||||
for e in "${all_entries[@]}"; do
|
||||
read -r kit verb <<< "$(marker_of "$e")"
|
||||
libs_into "$e"
|
||||
assemble "$e" "$profile" "$tmp/$kit.sh" ${libs[@]+"${libs[@]}"}
|
||||
done
|
||||
makefile "$tmp" "${all_entries[@]}"
|
||||
verify_kit "$tmp" "export" "${all_entries[@]}"
|
||||
|
||||
rm -rf "$dest"; mkdir -p "$(dirname "$dest")"; cp -r "$tmp" "$dest"
|
||||
echo " wrote $dest: $(cd "$dest" && ls | tr '\n' ' ')— verified to stand alone"
|
||||
if [ -n "${left// /}" ]; then
|
||||
echo
|
||||
echo " NOT carried — this machine's own, set them on the target if it needs them:"
|
||||
for k in $left; do echo " $k"; done
|
||||
fi
|
||||
;;
|
||||
*) echo "usage: $SELF_REL [write|check|export DIR]" >&2; exit 1 ;;
|
||||
esac
|
||||
@@ -1,24 +1,7 @@
|
||||
# Pinned toolchain — the single manifest ctrl/deps.sh installs from.
|
||||
# Every entry is a single binary; none of them needs an apt repo.
|
||||
# kubectl fully static
|
||||
# kind libc only
|
||||
# tilt libc + libstdc++ + libgcc (present in base Debian)
|
||||
# jq upstream static build (Debian's is linked against libjq/libonig)
|
||||
#
|
||||
# Checksums are the upstream-published SHA256 of the linux/amd64 artifact.
|
||||
#
|
||||
# To bump: change the version, then take the checksum from the release's own
|
||||
# published list — never hand-edit or hand-copy one from a download you did.
|
||||
# For anything hosted on GitHub releases that is:
|
||||
#
|
||||
# curl -sSL https://github.com/<org>/<repo>/releases/download/<tag>/checksums.txt \
|
||||
# | grep linux.x86_64
|
||||
#
|
||||
# (kubectl publishes its own instead: <KUBECTL_URL>.sha256.)
|
||||
#
|
||||
# There was a `ctrl/versions-refresh.sh` named here that has never existed. If
|
||||
# bumping stops being rare enough to do by hand, write it — but a comment
|
||||
# pointing at a missing script is worse than no comment.
|
||||
# Pinned toolchain (linux/amd64, upstream SHA256) — the manifest ctrl/deps.sh installs from.
|
||||
# To bump, take the checksum from the release's own list, e.g.
|
||||
# curl -sSL https://github.com/<org>/<repo>/releases/download/<tag>/checksums.txt | grep linux.x86_64
|
||||
# Notes: docs/notes/versions.md
|
||||
|
||||
KIND_VERSION=v0.32.0
|
||||
KIND_SHA256=50030de23cf40a18505f20426f6a8506bedf13c6e509244bd1fa9463721b0f54
|
||||
@@ -32,10 +15,7 @@ TILT_VERSION=0.37.6
|
||||
TILT_SHA256=e9672b8a18d43501f35dcfe98465969a7db0e436b36cf0c50c7e6f8d40de5fe6
|
||||
TILT_URL=https://github.com/tilt-dev/tilt/releases/download/v${TILT_VERSION}/tilt.${TILT_VERSION}.linux.x86_64.tar.gz
|
||||
|
||||
# ctlptl — creates a kind cluster WITH a local registry wired in, which is what
|
||||
# keeps images off docker.io (an unqualified name means docker.io/library/<name>).
|
||||
# Same publisher and same archive shape as tilt: binary at the archive root, so
|
||||
# fetch_tgz handles it with strip=0 and no special case.
|
||||
# ctlptl — kind cluster with a local registry wired in (keeps images off docker.io).
|
||||
CTLPTL_VERSION=0.9.4
|
||||
CTLPTL_SHA256=c63a1ec28e60bc3faf6becb76f53355c5cf5e0143dafdd27ad85db5584fa6b1e
|
||||
CTLPTL_URL=https://github.com/tilt-dev/ctlptl/releases/download/v${CTLPTL_VERSION}/ctlptl.${CTLPTL_VERSION}.linux.x86_64.tar.gz
|
||||
@@ -44,30 +24,34 @@ JQ_VERSION=1.8.2
|
||||
JQ_SHA256=b1c22172dd303f3be49e935aa56aa48a8b7a46e0bc838b4997d3bb451495870f
|
||||
JQ_URL=https://github.com/jqlang/jq/releases/download/jq-${JQ_VERSION}/jq-linux-amd64
|
||||
|
||||
# Node images shipped with KIND_VERSION above, pinned by digest so a kind upgrade
|
||||
# can never silently move the k8s version. Profiles select one via K8S_VERSION.
|
||||
# Older entries are kept deliberately: running a trailing-edge control plane is
|
||||
# part of simulating a legacy estate.
|
||||
# docker compose — often missing from distro packages; deps.sh links it into
|
||||
# ~/.docker/cli-plugins.
|
||||
COMPOSE_VERSION=5.5.1
|
||||
COMPOSE_SHA256=db1889184726840f75c4f9c001048430d4f25b3be3cb084d3ddd762bc0aed576
|
||||
COMPOSE_URL=https://github.com/docker/compose/releases/download/v${COMPOSE_VERSION}/docker-compose-linux-x86_64
|
||||
|
||||
# Node images for KIND_VERSION, pinned by digest; K8S_VERSION picks one (default: the newest).
|
||||
# Older entries are kept deliberately, for targets that run an older Kubernetes.
|
||||
NODE_IMAGE_v1_36=kindest/node:v1.36.1@sha256:3489c7674813ba5d8b1a9977baea8a6e553784dab7b84759d1014dbd78f7ebd5
|
||||
NODE_IMAGE_v1_35=kindest/node:v1.35.5@sha256:ce977ae6d65918d0b58a5f8b5e940429c2ce42fa3a5619ec2bbc60b949c0ac95
|
||||
NODE_IMAGE_v1_34=kindest/node:v1.34.8@sha256:02722c2dedddcfc00febf5d27fbeb9b7b2c14294c82109ff4a85d89ac9ba3256
|
||||
NODE_IMAGE_v1_33=kindest/node:v1.33.12@sha256:3f5c8443c620245e4d355cfe09e96a91ead32ceaa569d3f1ca9edf0cb2fe2ff4
|
||||
|
||||
# Images pulled at runtime (registry, mocks). Pinned by tag; the registry mode
|
||||
# decides where they are pulled FROM.
|
||||
# Images pulled at runtime. Pinned by tag; the registry mode decides where they are pulled FROM.
|
||||
REGISTRY_IMAGE=registry:2
|
||||
STUB_IMAGE=python:3.12-slim
|
||||
|
||||
# Addons, installed by ctrl/addons/<name>.sh when listed in a profile's ADDONS.
|
||||
# rig's own addons (ctrl/addons/<name>.sh), installed when ADDONS names them.
|
||||
CERT_MANAGER_VERSION=v1.21.1
|
||||
METRICS_SERVER_VERSION=v0.9.0
|
||||
METALLB_VERSION=v0.16.0
|
||||
|
||||
# Cabinets — public services dropped in as-is, the upstream image unmodified.
|
||||
# The same declaration installs on compose or in the cluster, so a dependency is
|
||||
# named once and works either way. Pinned by tag rather than
|
||||
# digest because they are ordinary upstream images with no supply chain claim
|
||||
# attached — bump freely, and preload them for the offline profile.
|
||||
POSTGRES_IMAGE=postgres:16-alpine
|
||||
REDIS_IMAGE=redis:7-alpine
|
||||
AIRFLOW_IMAGE=apache/airflow:2.10.4
|
||||
# The manifests those addons apply, fetched and verified like the binaries
|
||||
# (`deps.sh manifest <NAME>`), so an offline machine needs no network for them.
|
||||
# Sums from the release's own asset digest; metallb publishes none, see versions.md.
|
||||
CERT_MANAGER_MANIFEST_URL=https://github.com/cert-manager/cert-manager/releases/download/${CERT_MANAGER_VERSION}/cert-manager.yaml
|
||||
CERT_MANAGER_MANIFEST_SHA256=5f6a499b8c1857d57f560f536e0dcc830914b45c420899fe7ad0692c8624e408
|
||||
METRICS_SERVER_MANIFEST_URL=https://github.com/kubernetes-sigs/metrics-server/releases/download/${METRICS_SERVER_VERSION}/components.yaml
|
||||
METRICS_SERVER_MANIFEST_SHA256=1cec29a5267809306a2c6ec74a3e449abbb705b4a8beed0c8a1963910f72c79b
|
||||
METALLB_MANIFEST_URL=https://raw.githubusercontent.com/metallb/metallb/${METALLB_VERSION}/config/manifests/metallb-native.yaml
|
||||
METALLB_MANIFEST_SHA256=b0b9be2802f10aa32d45308b4457d06cde0c70544712c8d0cf5511657ffd2b69
|
||||
METALLB_MANIFEST_GIT_BLOB=7fbda334cc3ac0aaabdcb081af4f543feb3c2f9f
|
||||
|
||||
@@ -5,12 +5,12 @@ digraph rig_environment {
|
||||
node [fontname="Helvetica" fontsize=11 style=filled color="#1e2a4a" fontcolor="#e8eaf0" shape=box]
|
||||
edge [fontname="Helvetica" fontsize=9 fontcolor="#8892a8" color="#4a5568"]
|
||||
|
||||
label="One environment per directory — copies never collide"
|
||||
label="One environment per folder — copies never collide"
|
||||
labelloc=t
|
||||
fontsize=16
|
||||
fontcolor="#0066ff"
|
||||
|
||||
dirname [label="directory name\ne.g. acmebank/" fillcolor="#1f6feb" fontcolor="#ffffff" shape=octagon]
|
||||
dirname [label="folder name\nthe overlay's, else rig's\ne.g. platform-v2/" fillcolor="#1f6feb" fontcolor="#ffffff" shape=octagon]
|
||||
|
||||
subgraph cluster_derived {
|
||||
label="Everything below is derived from it"
|
||||
@@ -18,11 +18,11 @@ digraph rig_environment {
|
||||
color="#1e2a4a"
|
||||
fontcolor="#8892a8"
|
||||
|
||||
cname [label="cluster name\nacmebank" fillcolor="#121829"]
|
||||
ctx [label="kubectl context\nkind-acmebank" fillcolor="#121829"]
|
||||
img [label="image tag\nacmebank-deps" fillcolor="#121829"]
|
||||
ports [label="port block\n21300–21309" fillcolor="#121829"]
|
||||
reg [label="registry container\nacmebank-registry" fillcolor="#121829"]
|
||||
cname [label="cluster name\nplatform-v2" fillcolor="#121829"]
|
||||
ctx [label="kubectl context\nkind-platform-v2" fillcolor="#121829"]
|
||||
img [label="image tag\nplatform-v2-deps" fillcolor="#121829"]
|
||||
ports [label="port block\n20430–20439" fillcolor="#121829"]
|
||||
reg [label="registry container\nplatform-v2-registry" fillcolor="#121829"]
|
||||
}
|
||||
|
||||
subgraph cluster_config {
|
||||
@@ -32,9 +32,10 @@ digraph rig_environment {
|
||||
fontcolor="#8892a8"
|
||||
|
||||
versions [label="versions.env\npinned toolchain" fillcolor="#121829"]
|
||||
profile [label="env.d/<profile>.env\nnodes · CNI · audit · addons" fillcolor="#121829"]
|
||||
profile [label="env.d/<profile>.env\noptional: registry · mirror" fillcolor="#121829"]
|
||||
overlay [label="<overlay>/rig.env\noptional: what runs" fillcolor="#121829"]
|
||||
localenv [label="ctrl/.env\nsecrets, overrides" fillcolor="#121829"]
|
||||
shell [label="the environment\nPROFILE=client make …" fillcolor="#1a3a1a" fontcolor="#00c853"]
|
||||
shell [label="the environment\nOVERLAY=local/x make …" fillcolor="#1a3a1a" fontcolor="#00c853"]
|
||||
}
|
||||
|
||||
dirname -> cname
|
||||
@@ -44,7 +45,8 @@ digraph rig_environment {
|
||||
dirname -> reg
|
||||
|
||||
versions -> profile [label="overridden by"]
|
||||
profile -> localenv [label="overridden by"]
|
||||
profile -> overlay [label="overridden by"]
|
||||
overlay -> localenv [label="overridden by"]
|
||||
localenv -> shell [label="overridden by" color="#00c853"]
|
||||
|
||||
cluster [label="kind cluster" fillcolor="#1a1a3a" fontcolor="#0066ff" shape=octagon]
|
||||
|
||||
@@ -4,166 +4,181 @@
|
||||
<!-- Generated by graphviz version 14.1.2 (0)
|
||||
-->
|
||||
<!-- Title: rig_environment Pages: 1 -->
|
||||
<svg width="962pt" height="481pt"
|
||||
viewBox="0.00 0.00 962.00 481.00" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink">
|
||||
<g id="graph0" class="graph" transform="scale(1 1) rotate(0) translate(4 476.83)">
|
||||
<svg width="981pt" height="585pt"
|
||||
viewBox="0.00 0.00 981.00 585.00" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink">
|
||||
<g id="graph0" class="graph" transform="scale(1 1) rotate(0) translate(4 580.74)">
|
||||
<title>rig_environment</title>
|
||||
<polygon fill="#0a0e17" stroke="none" points="-4,4 -4,-476.83 958,-476.83 958,4 -4,4"/>
|
||||
<text xml:space="preserve" text-anchor="middle" x="477" y="-453.63" font-family="Helvetica,sans-Serif" font-size="16.00" fill="#0066ff">One environment per directory — copies never collide</text>
|
||||
<polygon fill="#0a0e17" stroke="none" points="-4,4 -4,-580.74 977,-580.74 977,4 -4,4"/>
|
||||
<text xml:space="preserve" text-anchor="middle" x="486.5" y="-557.54" font-family="Helvetica,sans-Serif" font-size="16.00" fill="#0066ff">One environment per folder — copies never collide</text>
|
||||
<g id="clust1" class="cluster">
|
||||
<title>cluster_derived</title>
|
||||
<polygon fill="#0a0e17" stroke="#1e2a4a" stroke-dasharray="5,2" points="8,-65 8,-144.5 596,-144.5 596,-65 8,-65"/>
|
||||
<text xml:space="preserve" text-anchor="middle" x="302" y="-125.3" font-family="Helvetica,sans-Serif" font-size="16.00" fill="#8892a8">Everything below is derived from it</text>
|
||||
<polygon fill="#0a0e17" stroke="#1e2a4a" stroke-dasharray="5,2" points="8,-65 8,-144.5 615,-144.5 615,-65 8,-65"/>
|
||||
<text xml:space="preserve" text-anchor="middle" x="311.5" y="-125.3" font-family="Helvetica,sans-Serif" font-size="16.00" fill="#8892a8">Everything below is derived from it</text>
|
||||
</g>
|
||||
<g id="clust2" class="cluster">
|
||||
<title>cluster_config</title>
|
||||
<polygon fill="#0a0e17" stroke="#1e2a4a" stroke-dasharray="5,2" points="604,-65 604,-437.33 946,-437.33 946,-65 604,-65"/>
|
||||
<text xml:space="preserve" text-anchor="middle" x="775" y="-418.13" font-family="Helvetica,sans-Serif" font-size="16.00" fill="#8892a8">Configuration — weakest first, later wins</text>
|
||||
<polygon fill="#0a0e17" stroke="#1e2a4a" stroke-dasharray="5,2" points="623,-65 623,-541.24 965,-541.24 965,-65 623,-65"/>
|
||||
<text xml:space="preserve" text-anchor="middle" x="794" y="-522.04" font-family="Helvetica,sans-Serif" font-size="16.00" fill="#8892a8">Configuration — weakest first, later wins</text>
|
||||
</g>
|
||||
<!-- dirname -->
|
||||
<g id="node1" class="node">
|
||||
<title>dirname</title>
|
||||
<polygon fill="#1f6feb" stroke="#1e2a4a" points="370.11,-197.44 370.11,-219.63 324.94,-235.33 261.06,-235.33 215.89,-219.63 215.89,-197.44 261.06,-181.75 324.94,-181.75 370.11,-197.44"/>
|
||||
<text xml:space="preserve" text-anchor="middle" x="293" y="-211.59" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#ffffff">directory name</text>
|
||||
<text xml:space="preserve" text-anchor="middle" x="293" y="-198.09" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#ffffff">e.g. acmebank/</text>
|
||||
<polygon fill="#1f6feb" stroke="#1e2a4a" points="411.98,-203.49 411.98,-234.25 346.97,-255.99 255.03,-255.99 190.02,-234.25 190.02,-203.49 255.03,-181.75 346.97,-181.75 411.98,-203.49"/>
|
||||
<text xml:space="preserve" text-anchor="middle" x="301" y="-228.67" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#ffffff">folder name</text>
|
||||
<text xml:space="preserve" text-anchor="middle" x="301" y="-215.17" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#ffffff">the overlay's, else rig's</text>
|
||||
<text xml:space="preserve" text-anchor="middle" x="301" y="-201.67" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#ffffff">e.g. platform-v2/</text>
|
||||
</g>
|
||||
<!-- cname -->
|
||||
<g id="node2" class="node">
|
||||
<title>cname</title>
|
||||
<polygon fill="#121829" stroke="#1e2a4a" points="104,-109 16,-109 16,-73 104,-73 104,-109"/>
|
||||
<text xml:space="preserve" text-anchor="middle" x="60" y="-94.05" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#e8eaf0">cluster name</text>
|
||||
<text xml:space="preserve" text-anchor="middle" x="60" y="-80.55" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#e8eaf0">acmebank</text>
|
||||
<text xml:space="preserve" text-anchor="middle" x="60" y="-80.55" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#e8eaf0">platform-v2</text>
|
||||
</g>
|
||||
<!-- dirname->cname -->
|
||||
<g id="edge1" class="edge">
|
||||
<title>dirname->cname</title>
|
||||
<path fill="none" stroke="#4a5568" d="M228.68,-192.51C192.88,-182.36 148.52,-166.7 113,-144.5 101.4,-137.25 90.34,-127.04 81.36,-117.55"/>
|
||||
<polygon fill="#4a5568" stroke="#4a5568" points="84.07,-115.32 74.76,-110.26 78.88,-120.02 84.07,-115.32"/>
|
||||
<path fill="none" stroke="#4a5568" d="M217.9,-193.66C183.86,-181.7 145.02,-165.31 113,-144.5 101.69,-137.15 90.82,-127.09 81.89,-117.75"/>
|
||||
<polygon fill="#4a5568" stroke="#4a5568" points="84.65,-115.58 75.31,-110.58 79.49,-120.31 84.65,-115.58"/>
|
||||
</g>
|
||||
<!-- ctx -->
|
||||
<g id="node3" class="node">
|
||||
<title>ctx</title>
|
||||
<polygon fill="#121829" stroke="#1e2a4a" points="223.75,-109 122.25,-109 122.25,-73 223.75,-73 223.75,-109"/>
|
||||
<text xml:space="preserve" text-anchor="middle" x="173" y="-94.05" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#e8eaf0">kubectl context</text>
|
||||
<text xml:space="preserve" text-anchor="middle" x="173" y="-80.55" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#e8eaf0">kind-acmebank</text>
|
||||
<polygon fill="#121829" stroke="#1e2a4a" points="228,-109 122,-109 122,-73 228,-73 228,-109"/>
|
||||
<text xml:space="preserve" text-anchor="middle" x="175" y="-94.05" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#e8eaf0">kubectl context</text>
|
||||
<text xml:space="preserve" text-anchor="middle" x="175" y="-80.55" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#e8eaf0">kind-platform-v2</text>
|
||||
</g>
|
||||
<!-- dirname->ctx -->
|
||||
<g id="edge2" class="edge">
|
||||
<title>dirname->ctx</title>
|
||||
<path fill="none" stroke="#4a5568" d="M265.77,-181.32C245.77,-162.07 218.77,-136.06 199.05,-117.08"/>
|
||||
<polygon fill="#4a5568" stroke="#4a5568" points="201.55,-114.63 191.91,-110.21 196.69,-119.67 201.55,-114.63"/>
|
||||
<path fill="none" stroke="#4a5568" d="M264.56,-181.46C244.01,-160.94 218.85,-135.8 200.43,-117.41"/>
|
||||
<polygon fill="#4a5568" stroke="#4a5568" points="203.11,-115.13 193.56,-110.54 198.16,-120.08 203.11,-115.13"/>
|
||||
</g>
|
||||
<!-- img -->
|
||||
<g id="node4" class="node">
|
||||
<title>img</title>
|
||||
<polygon fill="#121829" stroke="#1e2a4a" points="344.12,-109 241.88,-109 241.88,-73 344.12,-73 344.12,-109"/>
|
||||
<text xml:space="preserve" text-anchor="middle" x="293" y="-94.05" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#e8eaf0">image tag</text>
|
||||
<text xml:space="preserve" text-anchor="middle" x="293" y="-80.55" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#e8eaf0">acmebank-deps</text>
|
||||
<polygon fill="#121829" stroke="#1e2a4a" points="355.88,-109 246.12,-109 246.12,-73 355.88,-73 355.88,-109"/>
|
||||
<text xml:space="preserve" text-anchor="middle" x="301" y="-94.05" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#e8eaf0">image tag</text>
|
||||
<text xml:space="preserve" text-anchor="middle" x="301" y="-80.55" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#e8eaf0">platform-v2-deps</text>
|
||||
</g>
|
||||
<!-- dirname->img -->
|
||||
<g id="edge3" class="edge">
|
||||
<title>dirname->img</title>
|
||||
<path fill="none" stroke="#4a5568" d="M293,-181.32C293,-163.19 293,-139.07 293,-120.47"/>
|
||||
<polygon fill="#4a5568" stroke="#4a5568" points="296.5,-120.67 293,-110.67 289.5,-120.67 296.5,-120.67"/>
|
||||
<path fill="none" stroke="#4a5568" d="M301,-181.46C301,-162.23 301,-138.96 301,-120.97"/>
|
||||
<polygon fill="#4a5568" stroke="#4a5568" points="304.5,-120.98 301,-110.98 297.5,-120.98 304.5,-120.98"/>
|
||||
</g>
|
||||
<!-- ports -->
|
||||
<g id="node5" class="node">
|
||||
<title>ports</title>
|
||||
<polygon fill="#121829" stroke="#1e2a4a" points="451.38,-109 362.62,-109 362.62,-73 451.38,-73 451.38,-109"/>
|
||||
<text xml:space="preserve" text-anchor="middle" x="407" y="-94.05" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#e8eaf0">port block</text>
|
||||
<text xml:space="preserve" text-anchor="middle" x="407" y="-80.55" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#e8eaf0">21300–21309</text>
|
||||
<polygon fill="#121829" stroke="#1e2a4a" points="462.38,-109 373.62,-109 373.62,-73 462.38,-73 462.38,-109"/>
|
||||
<text xml:space="preserve" text-anchor="middle" x="418" y="-94.05" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#e8eaf0">port block</text>
|
||||
<text xml:space="preserve" text-anchor="middle" x="418" y="-80.55" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#e8eaf0">20430–20439</text>
|
||||
</g>
|
||||
<!-- dirname->ports -->
|
||||
<g id="edge4" class="edge">
|
||||
<title>dirname->ports</title>
|
||||
<path fill="none" stroke="#4a5568" d="M318.87,-181.32C337.78,-162.15 363.29,-136.3 382,-117.34"/>
|
||||
<polygon fill="#4a5568" stroke="#4a5568" points="384.47,-119.81 389,-110.24 379.49,-114.9 384.47,-119.81"/>
|
||||
<path fill="none" stroke="#4a5568" d="M334.84,-181.46C353.92,-160.94 377.28,-135.8 394.38,-117.41"/>
|
||||
<polygon fill="#4a5568" stroke="#4a5568" points="396.49,-120.29 400.73,-110.58 391.36,-115.52 396.49,-120.29"/>
|
||||
</g>
|
||||
<!-- reg -->
|
||||
<g id="node6" class="node">
|
||||
<title>reg</title>
|
||||
<polygon fill="#121829" stroke="#1e2a4a" points="588.38,-109 469.62,-109 469.62,-73 588.38,-73 588.38,-109"/>
|
||||
<text xml:space="preserve" text-anchor="middle" x="529" y="-94.05" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#e8eaf0">registry container</text>
|
||||
<text xml:space="preserve" text-anchor="middle" x="529" y="-80.55" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#e8eaf0">acmebank-registry</text>
|
||||
<polygon fill="#121829" stroke="#1e2a4a" points="607.12,-109 480.88,-109 480.88,-73 607.12,-73 607.12,-109"/>
|
||||
<text xml:space="preserve" text-anchor="middle" x="544" y="-94.05" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#e8eaf0">registry container</text>
|
||||
<text xml:space="preserve" text-anchor="middle" x="544" y="-80.55" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#e8eaf0">platform-v2-registry</text>
|
||||
</g>
|
||||
<!-- dirname->reg -->
|
||||
<g id="edge5" class="edge">
|
||||
<title>dirname->reg</title>
|
||||
<path fill="none" stroke="#4a5568" d="M350.86,-190.3C383.87,-179.35 425.43,-163.64 460,-144.5 474.27,-136.6 488.83,-125.98 500.87,-116.36"/>
|
||||
<polygon fill="#4a5568" stroke="#4a5568" points="502.85,-119.26 508.36,-110.21 498.41,-113.84 502.85,-119.26"/>
|
||||
<path fill="none" stroke="#4a5568" d="M373.79,-190.51C404.52,-177.93 440.23,-161.93 471,-144.5 485.65,-136.2 500.9,-125.58 513.66,-116.06"/>
|
||||
<polygon fill="#4a5568" stroke="#4a5568" points="515.41,-119.13 521.26,-110.3 511.18,-113.56 515.41,-119.13"/>
|
||||
</g>
|
||||
<!-- cluster -->
|
||||
<g id="node11" class="node">
|
||||
<g id="node12" class="node">
|
||||
<title>cluster</title>
|
||||
<polygon fill="#1a1a3a" stroke="#1e2a4a" points="460.81,-10.54 460.81,-25.46 429.29,-36 384.71,-36 353.19,-25.46 353.19,-10.54 384.71,0 429.29,0 460.81,-10.54"/>
|
||||
<text xml:space="preserve" text-anchor="middle" x="407" y="-14.3" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#0066ff">kind cluster</text>
|
||||
<polygon fill="#1a1a3a" stroke="#1e2a4a" points="471.81,-10.54 471.81,-25.46 440.29,-36 395.71,-36 364.19,-25.46 364.19,-10.54 395.71,0 440.29,0 471.81,-10.54"/>
|
||||
<text xml:space="preserve" text-anchor="middle" x="418" y="-14.3" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#0066ff">kind cluster</text>
|
||||
</g>
|
||||
<!-- cname->cluster -->
|
||||
<g id="edge9" class="edge">
|
||||
<g id="edge10" class="edge">
|
||||
<title>cname->cluster</title>
|
||||
<path fill="none" stroke="#4a5568" d="M93.35,-72.51C99.75,-69.67 106.48,-67 113,-65 189.55,-41.49 281.19,-29.58 341.58,-23.85"/>
|
||||
<polygon fill="#4a5568" stroke="#4a5568" points="341.71,-27.35 351.35,-22.96 341.07,-20.38 341.71,-27.35"/>
|
||||
<path fill="none" stroke="#4a5568" d="M93.06,-72.61C99.54,-69.72 106.38,-67.01 113,-65 193.43,-40.54 289.9,-28.78 352.5,-23.34"/>
|
||||
<polygon fill="#4a5568" stroke="#4a5568" points="352.6,-26.85 362.28,-22.53 352.02,-19.87 352.6,-26.85"/>
|
||||
</g>
|
||||
<!-- ports->cluster -->
|
||||
<g id="edge10" class="edge">
|
||||
<g id="edge11" class="edge">
|
||||
<title>ports->cluster</title>
|
||||
<path fill="none" stroke="#4a5568" d="M407,-72.81C407,-65.23 407,-56.1 407,-47.54"/>
|
||||
<polygon fill="#4a5568" stroke="#4a5568" points="410.5,-47.54 407,-37.54 403.5,-47.54 410.5,-47.54"/>
|
||||
<path fill="none" stroke="#4a5568" d="M418,-72.81C418,-65.23 418,-56.1 418,-47.54"/>
|
||||
<polygon fill="#4a5568" stroke="#4a5568" points="421.5,-47.54 418,-37.54 414.5,-47.54 421.5,-47.54"/>
|
||||
</g>
|
||||
<!-- versions -->
|
||||
<g id="node7" class="node">
|
||||
<title>versions</title>
|
||||
<polygon fill="#121829" stroke="#1e2a4a" points="750.38,-401.83 643.62,-401.83 643.62,-365.83 750.38,-365.83 750.38,-401.83"/>
|
||||
<text xml:space="preserve" text-anchor="middle" x="697" y="-386.88" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#e8eaf0">versions.env</text>
|
||||
<text xml:space="preserve" text-anchor="middle" x="697" y="-373.38" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#e8eaf0">pinned toolchain</text>
|
||||
<polygon fill="#121829" stroke="#1e2a4a" points="764.38,-505.74 657.62,-505.74 657.62,-469.74 764.38,-469.74 764.38,-505.74"/>
|
||||
<text xml:space="preserve" text-anchor="middle" x="711" y="-490.79" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#e8eaf0">versions.env</text>
|
||||
<text xml:space="preserve" text-anchor="middle" x="711" y="-477.29" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#e8eaf0">pinned toolchain</text>
|
||||
</g>
|
||||
<!-- profile -->
|
||||
<g id="node8" class="node">
|
||||
<title>profile</title>
|
||||
<polygon fill="#121829" stroke="#1e2a4a" points="781.5,-318.58 612.5,-318.58 612.5,-282.58 781.5,-282.58 781.5,-318.58"/>
|
||||
<text xml:space="preserve" text-anchor="middle" x="697" y="-303.63" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#e8eaf0">env.d/<profile>.env</text>
|
||||
<text xml:space="preserve" text-anchor="middle" x="697" y="-290.13" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#e8eaf0">nodes · CNI · audit · addons</text>
|
||||
<polygon fill="#121829" stroke="#1e2a4a" points="788.75,-422.49 633.25,-422.49 633.25,-386.49 788.75,-386.49 788.75,-422.49"/>
|
||||
<text xml:space="preserve" text-anchor="middle" x="711" y="-407.54" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#e8eaf0">env.d/<profile>.env</text>
|
||||
<text xml:space="preserve" text-anchor="middle" x="711" y="-394.04" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#e8eaf0">optional: registry · mirror</text>
|
||||
</g>
|
||||
<!-- versions->profile -->
|
||||
<g id="edge6" class="edge">
|
||||
<title>versions->profile</title>
|
||||
<path fill="none" stroke="#4a5568" d="M697,-365.59C697,-355.32 697,-342.03 697,-330.21"/>
|
||||
<polygon fill="#4a5568" stroke="#4a5568" points="700.5,-330.58 697,-320.58 693.5,-330.58 700.5,-330.58"/>
|
||||
<text xml:space="preserve" text-anchor="middle" x="728.5" y="-339.28" font-family="Helvetica,sans-Serif" font-size="9.00" fill="#8892a8">overridden by</text>
|
||||
<path fill="none" stroke="#4a5568" d="M711,-469.51C711,-459.24 711,-445.94 711,-434.13"/>
|
||||
<polygon fill="#4a5568" stroke="#4a5568" points="714.5,-434.49 711,-424.49 707.5,-434.49 714.5,-434.49"/>
|
||||
<text xml:space="preserve" text-anchor="middle" x="742.5" y="-443.19" font-family="Helvetica,sans-Serif" font-size="9.00" fill="#8892a8">overridden by</text>
|
||||
</g>
|
||||
<!-- localenv -->
|
||||
<!-- overlay -->
|
||||
<g id="node9" class="node">
|
||||
<title>localenv</title>
|
||||
<polygon fill="#121829" stroke="#1e2a4a" points="752.88,-226.54 637.12,-226.54 637.12,-190.54 752.88,-190.54 752.88,-226.54"/>
|
||||
<text xml:space="preserve" text-anchor="middle" x="695" y="-211.59" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#e8eaf0">ctrl/.env</text>
|
||||
<text xml:space="preserve" text-anchor="middle" x="695" y="-198.09" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#e8eaf0">secrets, overrides</text>
|
||||
<title>overlay</title>
|
||||
<polygon fill="#121829" stroke="#1e2a4a" points="772.25,-339.24 649.75,-339.24 649.75,-303.24 772.25,-303.24 772.25,-339.24"/>
|
||||
<text xml:space="preserve" text-anchor="middle" x="711" y="-324.29" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#e8eaf0"><overlay>/rig.env</text>
|
||||
<text xml:space="preserve" text-anchor="middle" x="711" y="-310.79" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#e8eaf0">optional: what runs</text>
|
||||
</g>
|
||||
<!-- profile->localenv -->
|
||||
<!-- profile->overlay -->
|
||||
<g id="edge7" class="edge">
|
||||
<title>profile->localenv</title>
|
||||
<path fill="none" stroke="#4a5568" d="M696.61,-282.22C696.34,-269.76 695.96,-252.69 695.64,-238.23"/>
|
||||
<polygon fill="#4a5568" stroke="#4a5568" points="699.14,-238.28 695.42,-228.36 692.14,-238.43 699.14,-238.28"/>
|
||||
<text xml:space="preserve" text-anchor="middle" x="727.68" y="-256.03" font-family="Helvetica,sans-Serif" font-size="9.00" fill="#8892a8">overridden by</text>
|
||||
<title>profile->overlay</title>
|
||||
<path fill="none" stroke="#4a5568" d="M711,-386.26C711,-375.99 711,-362.69 711,-350.88"/>
|
||||
<polygon fill="#4a5568" stroke="#4a5568" points="714.5,-351.24 711,-341.24 707.5,-351.24 714.5,-351.24"/>
|
||||
<text xml:space="preserve" text-anchor="middle" x="742.5" y="-359.94" font-family="Helvetica,sans-Serif" font-size="9.00" fill="#8892a8">overridden by</text>
|
||||
</g>
|
||||
<!-- shell -->
|
||||
<!-- localenv -->
|
||||
<g id="node10" class="node">
|
||||
<title>localenv</title>
|
||||
<polygon fill="#121829" stroke="#1e2a4a" points="768.88,-236.87 653.12,-236.87 653.12,-200.87 768.88,-200.87 768.88,-236.87"/>
|
||||
<text xml:space="preserve" text-anchor="middle" x="711" y="-221.92" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#e8eaf0">ctrl/.env</text>
|
||||
<text xml:space="preserve" text-anchor="middle" x="711" y="-208.42" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#e8eaf0">secrets, overrides</text>
|
||||
</g>
|
||||
<!-- overlay->localenv -->
|
||||
<g id="edge8" class="edge">
|
||||
<title>overlay->localenv</title>
|
||||
<path fill="none" stroke="#4a5568" d="M711,-302.76C711,-287.79 711,-265.95 711,-248.44"/>
|
||||
<polygon fill="#4a5568" stroke="#4a5568" points="714.5,-248.67 711,-238.67 707.5,-248.67 714.5,-248.67"/>
|
||||
<text xml:space="preserve" text-anchor="middle" x="742.5" y="-276.69" font-family="Helvetica,sans-Serif" font-size="9.00" fill="#8892a8">overridden by</text>
|
||||
</g>
|
||||
<!-- shell -->
|
||||
<g id="node11" class="node">
|
||||
<title>shell</title>
|
||||
<polygon fill="#1a3a1a" stroke="#1e2a4a" points="763.38,-109 614.62,-109 614.62,-73 763.38,-73 763.38,-109"/>
|
||||
<text xml:space="preserve" text-anchor="middle" x="689" y="-94.05" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#00c853">the environment</text>
|
||||
<text xml:space="preserve" text-anchor="middle" x="689" y="-80.55" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#00c853">PROFILE=client make …</text>
|
||||
<polygon fill="#1a3a1a" stroke="#1e2a4a" points="791,-109 631,-109 631,-73 791,-73 791,-109"/>
|
||||
<text xml:space="preserve" text-anchor="middle" x="711" y="-94.05" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#00c853">the environment</text>
|
||||
<text xml:space="preserve" text-anchor="middle" x="711" y="-80.55" font-family="Helvetica,sans-Serif" font-size="11.00" fill="#00c853">OVERLAY=local/x make …</text>
|
||||
</g>
|
||||
<!-- localenv->shell -->
|
||||
<g id="edge8" class="edge">
|
||||
<g id="edge9" class="edge">
|
||||
<title>localenv->shell</title>
|
||||
<path fill="none" stroke="#00c853" d="M694.11,-190.49C693.16,-172.16 691.63,-142.72 690.49,-120.79"/>
|
||||
<polygon fill="#00c853" stroke="#00c853" points="694,-120.81 689.99,-111.01 687.01,-121.18 694,-120.81"/>
|
||||
<text xml:space="preserve" text-anchor="middle" x="724.21" y="-155.2" font-family="Helvetica,sans-Serif" font-size="9.00" fill="#8892a8">overridden by</text>
|
||||
<path fill="none" stroke="#00c853" d="M711,-200.63C711,-180.03 711,-145.27 711,-120.62"/>
|
||||
<polygon fill="#00c853" stroke="#00c853" points="714.5,-120.95 711,-110.95 707.5,-120.95 714.5,-120.95"/>
|
||||
<text xml:space="preserve" text-anchor="middle" x="742.5" y="-155.2" font-family="Helvetica,sans-Serif" font-size="9.00" fill="#8892a8">overridden by</text>
|
||||
</g>
|
||||
<!-- shell->cluster -->
|
||||
<g id="edge11" class="edge">
|
||||
<g id="edge12" class="edge">
|
||||
<title>shell->cluster</title>
|
||||
<path fill="none" stroke="#4a5568" stroke-dasharray="5,2" d="M628.29,-72.6C618.83,-69.99 609.17,-67.38 600,-65 553.72,-52.99 500.89,-40.48 462.22,-31.54"/>
|
||||
<polygon fill="#4a5568" stroke="#4a5568" points="463.2,-28.18 452.67,-29.35 461.63,-35 463.2,-28.18"/>
|
||||
<path fill="none" stroke="#4a5568" stroke-dasharray="5,2" d="M648.24,-72.58C638.47,-69.98 628.47,-67.37 619,-65 570.38,-52.84 514.82,-40.22 474.45,-31.28"/>
|
||||
<polygon fill="#4a5568" stroke="#4a5568" points="475.23,-27.87 464.71,-29.13 473.72,-34.7 475.23,-27.87"/>
|
||||
</g>
|
||||
</g>
|
||||
</svg>
|
||||
|
||||
|
Before Width: | Height: | Size: 11 KiB After Width: | Height: | Size: 12 KiB |
@@ -240,6 +240,7 @@
|
||||
<a href="#steps">The steps</a>
|
||||
<a href="#install">Installation</a>
|
||||
<a href="#environments">Environments</a>
|
||||
<a href="#overlays">Overlays</a>
|
||||
<a href="#profiles">Profiles</a>
|
||||
<a href="#registry">Registry</a>
|
||||
<a href="#architecture">Architecture</a>
|
||||
@@ -267,7 +268,7 @@
|
||||
<pre><code><span class="c"># then, in the environment directory:</span>
|
||||
make check <span class="c"># is this machine ready? reports, never fixes</span>
|
||||
make deps <span class="c"># install the pinned toolchain</span>
|
||||
make cluster up <span class="c"># build the cluster for the active profile</span>
|
||||
make cluster up <span class="c"># cluster + registry + addons; ports derive by themselves</span>
|
||||
</code></pre>
|
||||
<p>Read <code>make check</code> before <code>make deps</code>. It never changes
|
||||
anything — it prints what it found and, at the end, the steps it cannot perform
|
||||
@@ -288,30 +289,24 @@ make cluster up <span class="c"># build the cluster for the active profile</spa
|
||||
one failure at a time.</p>
|
||||
<pre><code>make check</code></pre>
|
||||
|
||||
<h3>2 · make setup</h3>
|
||||
<p>Does the preparation that can be automated: installs the pinned
|
||||
toolchain if it is missing, checks PATH, Docker, and this environment's
|
||||
ports. Every step is independently checked, so running it twice is safe and
|
||||
running it half-configured finishes the job.</p>
|
||||
<p>It <b>does not stop at the first failure</b>. A setup script that dies at
|
||||
step two hides that steps four and five would also have failed, and on an
|
||||
unfamiliar machine the complete list is the point. The tail of the output is
|
||||
a to-do list of only what is outstanding.</p>
|
||||
<pre><code>make setup <span class="c"># host + toolchain</span>
|
||||
make setup --share-docker <span class="c"># ...and offer this machine's Docker to other distros</span>
|
||||
</code></pre>
|
||||
<h3>2 · make deps</h3>
|
||||
<p>Installs the pinned toolchain — only what is missing — and tells you
|
||||
if its directory is not on PATH yet. Running it twice is safe.</p>
|
||||
<pre><code>make deps</code></pre>
|
||||
|
||||
<h3>3 · make cluster up</h3>
|
||||
<p>Builds the cluster for the active profile. It prints what the profile
|
||||
locks in <i>before</i> spending the time, because the CNI and the audit
|
||||
policy are fixed at creation and cannot be changed afterwards.</p>
|
||||
<p>Builds the cluster, starts its registry and installs the profile's
|
||||
addons — there is nothing else to run first. It prints what the profile
|
||||
locks in <i>before</i> spending the time, because the kind config is
|
||||
fixed at creation and cannot be changed afterwards.</p>
|
||||
<p>Re-running is safe and, more importantly, <b>convergent</b>: if a first
|
||||
attempt was interrupted before the CNI was installed, running it again
|
||||
finishes the job rather than reporting "already exists" and leaving every
|
||||
node permanently NotReady.</p>
|
||||
<pre><code>make cluster up <span class="c"># default profile</span>
|
||||
make cluster up PROFILE=client <span class="c"># three nodes, audit on, cached registry</span>
|
||||
make cluster reset <span class="c"># destroy and rebuild — the only way to change CNI or audit</span>
|
||||
<pre><code>make cluster up <span class="c"># built-in defaults — no profile needed</span>
|
||||
make cluster up PROFILE=mirror <span class="c"># after copying env.d/mirror.env.example: cached registry</span>
|
||||
make cluster up OVERLAY=examples/data <span class="c"># an overlay: what runs, kept outside rig</span>
|
||||
make cluster reset <span class="c"># destroy and rebuild — how an edited kind config takes effect</span>
|
||||
</code></pre>
|
||||
|
||||
<h3>4 · make docs</h3>
|
||||
@@ -323,18 +318,17 @@ make cluster reset <span class="c"># destroy and rebuild — the on
|
||||
<h3>Checking on things</h3>
|
||||
<dl>
|
||||
<dt>make cluster list</dt><dd>Every cluster on the machine, its memory cost and its port block. The usual reason a new one will not start is an old one you forgot about; <code>make cluster free</code> frees them without deleting.</dd>
|
||||
<dt>make ports</dt><dd>This environment's port block, and whether each is derived or overridden.</dd>
|
||||
<dt>make registry</dt><dd>Which of the four registry modes is active, and where it points.</dd>
|
||||
<dt>make dockerhost</dt><dd>Which WSL distro owns Docker and what this one is using.</dd>
|
||||
<dt>make check</dt><dd>Short: host, toolchain, and whether this cluster fits, its ports, registry and addons. Details only appear when something needs attention; <code>make check all</code> prints every one.</dd>
|
||||
<dt>make check mem</dt><dd>Memory in depth: what caps it, how far it really climbs, and on WSL the <code>.wslconfig</code> backup and restore.</dd>
|
||||
</dl>
|
||||
|
||||
<h3>Running more than one</h3>
|
||||
<p>Copy the directory, rename it, and repeat from step 2. Cluster name,
|
||||
context, image tags and the port block all follow the directory name, so
|
||||
the second environment collides with nothing and neither one's teardown can
|
||||
reach the other.</p>
|
||||
<pre><code>cp -r rig ../platform-v2 && cd ../platform-v2
|
||||
make setup && make cluster up
|
||||
<p>Name another overlay, or copy the directory and rename it. Cluster name,
|
||||
context, image tags and the port block all follow the folder name — the
|
||||
overlay's, or rig's — so the second environment collides with nothing and
|
||||
neither one's teardown can reach the other.</p>
|
||||
<pre><code>OVERLAY=local/platform-v2 make cluster up
|
||||
cp -r rig ../platform-v3 && cd ../platform-v3 && make cluster up
|
||||
</code></pre>
|
||||
</div>
|
||||
</section>
|
||||
@@ -372,65 +366,90 @@ make setup && make cluster up
|
||||
</table>
|
||||
<pre><code>make deps core <span class="c"># kubectl and jq only — nothing that creates a cluster</span>
|
||||
make deps <span class="c"># dev, the default</span>
|
||||
make setup core <span class="c"># same distinction, via setup</span>
|
||||
</code></pre>
|
||||
<p>Testing <i>in situ</i> on a managed machine is still possible — install
|
||||
the <code>dev</code> tier deliberately when you need it. The point is that
|
||||
it should be a decision rather than a side effect of running setup.</p>
|
||||
it should be a decision rather than a side effect of installing.</p>
|
||||
<p>The documentation itself needs neither tier: <code>make docs</code>
|
||||
wants only Docker.</p>
|
||||
|
||||
<h3>Air-gapped</h3>
|
||||
<pre><code>make deps-image full <span class="c"># bakes every binary into the image</span>
|
||||
<pre><code>make deps image full <span class="c"># bakes every binary into the image</span>
|
||||
docker save …-deps:full | gzip > rig.tgz
|
||||
<span class="c"># carry that one file in, then:</span>
|
||||
docker load < rig.tgz && make cluster up PROFILE=offline
|
||||
docker load < rig.tgz && make cluster up PROFILE=offline <span class="c"># from env.d/offline.env.example</span>
|
||||
</code></pre>
|
||||
</div>
|
||||
</section>
|
||||
|
||||
<section class="section" id="environments">
|
||||
<h2>Environments</h2>
|
||||
<p class="lede">One directory is one environment. Copy it, rename it, run it.</p>
|
||||
<p class="lede">One folder is one environment — an overlay's, or rig's own. Copy it, rename it, run it.</p>
|
||||
<div class="graph-container">
|
||||
<a href="viewer.html?src=graphs/02-environment.svg"><img src="graphs/02-environment.svg" alt="Environment derivation"></a>
|
||||
</div>
|
||||
<div class="prose">
|
||||
<p>Running several versions of a system at once means several clusters on one
|
||||
machine, not several machines. Everything that could collide is derived from
|
||||
the directory name:</p>
|
||||
the folder name (the overlay's when one is named):</p>
|
||||
<dl>
|
||||
<dt>cluster + context</dt><dd><code>acmebank/</code> builds <code>acmebank</code> on <code>kind-acmebank</code>.</dd>
|
||||
<dt>cluster + context</dt><dd><code>platform-v2/</code> builds <code>platform-v2</code> on <code>kind-platform-v2</code>.</dd>
|
||||
<dt>port block</dt><dd>Ten ports from a hash of the name, in the 20000+ range — clear of 80, 443, 3000, 5432, 8000 and 8080.</dd>
|
||||
<dt>registry + images</dt><dd>Named after the environment, so two copies never share one.</dd>
|
||||
</dl>
|
||||
<p>Two copies therefore never collide, and neither one's
|
||||
<code>make cluster down</code> can touch the other. <code>make ports</code>
|
||||
shows the block; <code>make ports persist</code> freezes it into
|
||||
<code>make cluster down</code> can touch the other. <code>make check</code>
|
||||
shows the block; <code>bash ctrl/ports.sh persist</code> freezes it into
|
||||
<code>ctrl/.env</code> if you want it fixed rather than derived.</p>
|
||||
|
||||
<h3>Configuration layers</h3>
|
||||
<p>Weakest first, later wins: pinned versions → the profile →
|
||||
<p>Weakest first, later wins: built-in defaults → pinned versions → a
|
||||
profile, if you name one → the overlay's <code>rig.env</code> →
|
||||
<code>ctrl/.env</code> → the environment. So
|
||||
<code>make cluster up PROFILE=client</code> always beats every file.</p>
|
||||
<code>make cluster up PROFILE=<name></code> always beats every file.</p>
|
||||
</div>
|
||||
</section>
|
||||
|
||||
<section class="section" id="overlays">
|
||||
<h2>Overlays</h2>
|
||||
<p class="lede">What runs lives outside rig — rig reads it and never writes into it.</p>
|
||||
<div class="prose">
|
||||
<p>An overlay is one folder, kept outside rig's version control, holding a
|
||||
use case. Every piece is optional:</p>
|
||||
<table>
|
||||
<tr><th>in the overlay</th><th>what rig does with it</th></tr>
|
||||
<tr><td><code>rig.env</code></td><td>a config layer: addons, namespaces, images — anything a profile could set</td></tr>
|
||||
<tr><td><code>k8s/overlays/dev/</code></td><td>the manifests the dev loop applies</td></tr>
|
||||
<tr><td><code>kind-config.yaml.tpl</code></td><td>the cluster's shape, when it needs its own (mounts, ports)</td></tr>
|
||||
<tr><td><code>addons/<name>.sh</code></td><td>addons, found before rig's own</td></tr>
|
||||
<tr><td><code>Tiltfile</code></td><td>the workload's half of the dev loop, included by rig's</td></tr>
|
||||
</table>
|
||||
<pre><code>cp -r examples/starter local/myenv <span class="c"># local/ is gitignored</span>
|
||||
OVERLAY=local/myenv make cluster up
|
||||
</code></pre>
|
||||
<p>With none named, rig runs its own <code>examples/starter</code>.
|
||||
<code>examples/data</code> carries postgres, redis and airflow as an
|
||||
overlay's own addons. A project can also carry rig at
|
||||
<code><project>/rig/</code> and be the overlay itself, with a
|
||||
three-line forwarding Makefile — see <code>docs/notes/overlay.md</code>.</p>
|
||||
</div>
|
||||
</section>
|
||||
|
||||
<section class="section" id="profiles">
|
||||
<h2>Profiles</h2>
|
||||
<p class="lede">Cluster shape is declared, not baked in.</p>
|
||||
<p class="lede">How this machine reaches the world — optional; rig needs none.</p>
|
||||
<div class="prose">
|
||||
<table>
|
||||
<tr><th>profile</th><th>nodes</th><th>audit</th><th>registry</th><th>for</th></tr>
|
||||
<tr><td><code>minimal</code></td><td>1</td><td>off</td><td>none</td><td>first boot; assumes nothing</td></tr>
|
||||
<tr><td><code>client</code></td><td>3</td><td>on</td><td>mirror</td><td>the regulated shape</td></tr>
|
||||
<tr><td><code>offline</code></td><td>1</td><td>on</td><td>local</td><td>air-gapped</td></tr>
|
||||
<tr><th>example</th><th>registry</th><th>for</th></tr>
|
||||
<tr><td><i>none</i></td><td>local</td><td>the built-in defaults; no profile needed</td></tr>
|
||||
<tr><td><code>mirror.env.example</code></td><td>mirror</td><td>images through an internal registry</td></tr>
|
||||
<tr><td><code>offline.env.example</code></td><td>local</td><td>air-gapped</td></tr>
|
||||
</table>
|
||||
|
||||
<div class="note"><p><b>The audit policy cannot be changed later.</b> It is an
|
||||
apiserver flag, fixed when the cluster is created. <code>cluster up</code>
|
||||
prints what a profile locks in before spending the time, and
|
||||
<code>make cluster reset</code> is the way out.</p></div>
|
||||
<div class="note"><p><b>The kind config cannot be re-applied.</b> Edit
|
||||
<code>ctrl/k8s/kind-config.yaml.tpl</code> (or the overlay's own); it takes effect when the cluster is created. <code>cluster up</code> prints what it
|
||||
locks in before spending the time, and <code>make cluster reset</code> is
|
||||
the way out.</p></div>
|
||||
|
||||
<h3>LoadBalancer services</h3>
|
||||
<p>Real manifests use <code>type: LoadBalancer</code>, because a real
|
||||
|
||||
34
rig/docs/notes/Dockerfile.deps.md
Normal file
34
rig/docs/notes/Dockerfile.deps.md
Normal file
@@ -0,0 +1,34 @@
|
||||
# ctrl/Dockerfile.deps
|
||||
|
||||
## Purpose
|
||||
|
||||
The toolchain installer image. It does NOT run the cluster — it installs a toolchain onto the host and gets out of the way.
|
||||
|
||||
This exists to kill a bootstrap paradox: a plain bash installer needs curl, jq and sha256sum to already be present, and a minimal Debian has none of them. It carries its own toolchain, so the only host prerequisite is Docker.
|
||||
|
||||
## Variants
|
||||
|
||||
Two variants from one file:
|
||||
|
||||
```
|
||||
docker build -f ctrl/Dockerfile.deps --target deps -t <slug>-deps .
|
||||
docker build -f ctrl/Dockerfile.deps --target deps-full -t <slug>-deps:full .
|
||||
```
|
||||
|
||||
`deps-full` bakes every pinned binary in at build time, and the manifests rig's own addons apply (metallb, cert-manager, metrics-server). `docker save` it and you have the whole installer as one file to carry into an air-gapped network. There, put the manifests where the addons look for them:
|
||||
|
||||
```
|
||||
docker run --rm -v "$PWD/vendor:/out/vendor" rig-deps:full manifests --to /out/vendor/manifests
|
||||
```
|
||||
|
||||
Each is verified against its pin on the way out, and again when an addon uses it. The addons' container images still have to be preloaded into the local registry: the manifests reference quay.io and registry.k8s.io, and registry mirroring covers docker.io only.
|
||||
|
||||
## Packages
|
||||
|
||||
ca-certificates + curl: fetch and verify. graphviz + python3: render diagrams and validate the arch model, so the host never needs an apt package.
|
||||
|
||||
docker-cli, NOT docker.io: we only ever talk to the host's daemon through the mounted socket, and under `--no-install-recommends` the docker.io package ships docker-init without the actual `docker` binary.
|
||||
|
||||
## The installer is the standalone kit
|
||||
|
||||
The installer is the generated standalone kit, not deps.sh plus the files it reads. A kit is one file with its pins frozen in and is proven to run with nothing else from rig present — which is exactly what an image needs, and `make standalone` keeps it current. Pins are the same in every profile's kit.
|
||||
45
rig/docs/notes/Dockerfile.example.md
Normal file
45
rig/docs/notes/Dockerfile.example.md
Normal file
@@ -0,0 +1,45 @@
|
||||
# examples/starter/Dockerfile.example
|
||||
|
||||
## Naming
|
||||
|
||||
EXAMPLE — a component image. Copy, rename, replace. Named like the manifest it feeds and the resource it becomes:
|
||||
|
||||
```
|
||||
Dockerfile.api -> image <cluster>-api -> image: in k8s/base/api.yaml
|
||||
```
|
||||
|
||||
That image string is the ONLY thing connecting the three. Nothing checks it; a typo shows up as a pod stuck in ImagePullBackOff pulling from the public index, which reads like a network problem and is not one.
|
||||
|
||||
## COPY paths are relative to the build context
|
||||
|
||||
The overlay's Tiltfile runs from the overlay's own folder (rig's ctrl/Tiltfile includes it), so both paths in `docker_build` are relative to the overlay:
|
||||
|
||||
```
|
||||
context='.' the overlay folder (or a subfolder, e.g. 'repodir/api')
|
||||
dockerfile='Dockerfile.api' relative to the overlay's Tiltfile too
|
||||
```
|
||||
|
||||
Every COPY is resolved against the context, NOT against the Dockerfile's directory. With `context='.'` and the Dockerfile in a subfolder, a file sitting right beside it is still reached through that subfolder:
|
||||
|
||||
```
|
||||
COPY docker/nginx.conf /etc/nginx/conf.d/default.conf # correct, Dockerfile in docker/
|
||||
COPY nginx.conf /etc/nginx/conf.d/default.conf # fails — no such file in the context
|
||||
```
|
||||
|
||||
Nothing warns you. The build just cannot find a file that is visibly there.
|
||||
|
||||
Before overlays, rig's own ctrl/Tiltfile built with `context='..'` (the repository root) and the Dockerfile in ctrl/, which is the same trap one level up.
|
||||
|
||||
## Dependency layer
|
||||
|
||||
Dependencies first, in their own layer: they change far less often than the code, so a source edit does not reinstall them on every rebuild.
|
||||
|
||||
## live_update
|
||||
|
||||
The sync in the Tiltfile's `docker_build` must land where this image expects it:
|
||||
|
||||
```
|
||||
live_update=[sync('api', '/app/api')]
|
||||
```
|
||||
|
||||
matches `COPY api/ ./api/` with `WORKDIR /app`. If the two disagree, Tilt syncs into a path nothing reads and the container keeps serving the built copy — edits appear to do nothing, with no error anywhere.
|
||||
58
rig/docs/notes/Makefile.md
Normal file
58
rig/docs/notes/Makefile.md
Normal file
@@ -0,0 +1,58 @@
|
||||
# Makefile
|
||||
|
||||
## Shape and config layers
|
||||
|
||||
Thin control Makefile: few targets, and the subcommand is an argument rather than a second target: `make cluster down`, not `make cluster-down`.
|
||||
|
||||
```
|
||||
make check is this machine ready? (never changes anything)
|
||||
make deps install the toolchain
|
||||
make cluster up cluster + registry + addons (ports derive by themselves)
|
||||
make tilt / docs work on it, read about it
|
||||
```
|
||||
|
||||
The logic lives in the scripts, never here: `make cluster up` -> ctrl/cluster.sh up.
|
||||
|
||||
Config layers, weakest first: built-in defaults < ctrl/versions.env (pinned toolchain) < ctrl/env.d/<profile>.env (optional) < <overlay>/rig.env (optional) < ctrl/.env (local, gitignored) < the environment. So `make cluster up PROFILE=<name>` beats them all. See [config.md](config.md) and [overlay.md](overlay.md).
|
||||
|
||||
Start with: `make check && make deps && make cluster up`
|
||||
|
||||
## FACTS
|
||||
|
||||
Identity follows the FOLDER NAME — the overlay's when one is named, else this directory's — so either can be copied elsewhere, renamed, and run as a separate environment with no edits. ctrl/.env overrides it when you want a name that differs from the folder.
|
||||
|
||||
Asked once, of ctrl/ports.sh, which resolves it through lib/config.sh:
|
||||
|
||||
```
|
||||
CLUSTER KUBECONTEXT HTTP HTTPS TILT REGISTRY MANIFESTS_DIR OVERLAY_DIR
|
||||
```
|
||||
|
||||
Read positionally, so the order is a contract; ctrl/selftest.sh pins it. The two paths are absolute, or `-` when there is none, so the count never shifts.
|
||||
|
||||
`OVERLAY` and `CLUSTER` given as make arguments (`make tilt OVERLAY=local/x`) are handed to that `$(shell ...)` explicitly. Before make 4.4, `$(shell)` runs with make's own environment and does not see command-line variables, while the recipes do: tilt would then be told one context and the Tiltfile would guard on another. An overlay's forwarder avoids the question by putting `OVERLAY` in the environment.
|
||||
|
||||
This used to be sed over ctrl/.env plus a slug computed in the Makefile, which is a SECOND derivation of values lib/config.sh already owns, and the two could disagree about the port after `ports.sh persist`, or about the name for any directory whose sanitised form differs from its raw one. One source now; the Tiltfile reads the same line.
|
||||
|
||||
## CLUSTER / KCTX fallback
|
||||
|
||||
The fallback matters: ports.sh sources config.sh, and if a profile or .env is broken it exits non-zero. Losing the cluster name would send --context to the wrong place, so fall back to the folder rather than to empty.
|
||||
|
||||
## ARGS as .PHONY
|
||||
|
||||
Words after the target become the script's subcommand; each gets a no-op rule so make does not treat them as goals. They are also marked PHONY, because some of those words name real directories. `cfg`, `ctrl`, `docs`, `gen` and `init` all exist at this level, and make considers a target that is an existing directory already built, so `make build ctrl` ran the build and then printed "make: 'ctrl' is up to date". The empty rule is not enough on its own; only .PHONY stops make consulting the filesystem.
|
||||
|
||||
## tilt: --port guard
|
||||
|
||||
--port is only passed when TILT_PORT resolved. It normally does, since FACTS asks ports.sh, but ports.sh can fail on a broken profile, and without the guard tilt receives a bare `--port` with no value and fails on the flag rather than on anything real. Tilt's own default is 10350, which is the number every project on this machine is trying not to collide on, so falling back to it silently is worse than not passing the flag.
|
||||
|
||||
The Tiltfile asks ports.sh for the rest itself (cluster, registry and where the manifests are), so nothing needs passing here beyond what tilt's own flags require.
|
||||
|
||||
## Aliases (kind-up, tilt-up, ...)
|
||||
|
||||
Aliases, not a second implementation: each one calls the same script the canonical target does.
|
||||
|
||||
The header argues for `make cluster down` over `make cluster-down`, and that still holds *within* the Makefile. But rig is one repo among several on the same machine, and every other one answers to kind-up / tilt-up. Muscle memory spanning six projects beats internal tidiness in one, so both spellings work.
|
||||
|
||||
`cluster list` and `cluster free` have no hyphenated twin on purpose: they are rig's own, with nothing to be consistent with.
|
||||
|
||||
Nothing outside the Makefile reads these names: the script is `ctrl/cluster.sh` and it takes the verb. So rename them, delete the ones you never type, or add the spelling your own projects use. An alias is two lines, and adding one costs nothing but a line in .PHONY.
|
||||
43
rig/docs/notes/Tiltfile.md
Normal file
43
rig/docs/notes/Tiltfile.md
Normal file
@@ -0,0 +1,43 @@
|
||||
# ctrl/Tiltfile
|
||||
|
||||
## Purpose and ownership
|
||||
|
||||
rig's half of the dev loop, and rig's file: who we are, the context guard, the registry, the overlay's manifests and the namespaces they use. The workload's half — images, resource names and order, port-forwards — is the overlay's own `Tiltfile`, which this one includes at the end (see [overlay.md](overlay.md)). With no overlay named that is `examples/starter/Tiltfile`, so `make tilt` on a fresh clone comes up with the two examples running and nothing to edit.
|
||||
|
||||
Splitting it is what lets rig be replaced as a whole (✖ S1 in STALE.md).
|
||||
|
||||
## Nothing hardcoded to this directory
|
||||
|
||||
Nothing in the Tiltfile is hardcoded to this directory, deliberately. Every other project here writes its slug into the Tiltfile five or six times by hand, so a copy of the project deploys into the original's cluster until someone remembers to edit all of them. A rig is meant to be copied and renamed, and an overlay moved, so it asks instead.
|
||||
|
||||
## Who we are, and on which ports
|
||||
|
||||
One question to rig, answered by ctrl/ports.sh, which resolves it through lib/config.sh — the same path every other rig script takes. That is the point: the cluster name is NOT the bare directory name (it is lowercased and reduced to a DNS label), and the ports honour anything pinned in ctrl/.env. Recomputing either of those here in Starlark is how two copies end up disagreeing about which cluster they are talking to.
|
||||
|
||||
The manifests and the overlay arrive as absolute paths, or `-` when there is none.
|
||||
|
||||
## Refuse to deploy into the wrong cluster
|
||||
|
||||
Tilt snapshots the kubectl context at startup, BEFORE parsing this file, so it cannot be switched from here — only refused. `make tilt` passes --context for you; the guard catches a bare `tilt up` after some other project moved the global context.
|
||||
|
||||
## Images go to this environment's own registry
|
||||
|
||||
Fail closed. Tilt can usually infer the kind registry on its own, but "usually" is an inference, and when it misses, an unqualified name like `app` quietly means docker.io/library/app — a push to the public index instead of the registry two lines away. rig runs that registry; name it.
|
||||
|
||||
## Namespaces
|
||||
|
||||
Every namespace the manifests use has to exist before anything lands in it, and kustomize does not guarantee ordering across resources, so the Tiltfile creates them first (idempotent). The Namespaces the manifests declare are grouped as the `infra` resource, whatever they are called.
|
||||
|
||||
Nothing here assumes a namespace is named after the cluster (✖ S4).
|
||||
|
||||
## Handing over to the overlay
|
||||
|
||||
The facts are published as environment variables (`os.putenv`) and the overlay's Tiltfile is `include()`d. An included Tiltfile runs from its own folder: `os.getcwd()`, `local()` and every relative path in it resolve from the overlay, so it needs no path back into rig and reads the facts with `os.getenv`:
|
||||
|
||||
```
|
||||
RIG_CLUSTER RIG_CONTEXT RIG_HTTP_PORT RIG_HTTPS_PORT RIG_TILT_PORT RIG_REGISTRY RIG_OVERLAY_DIR
|
||||
```
|
||||
|
||||
## Catalogue
|
||||
|
||||
The blocks that recur across projects moved with the workload's half: `examples/starter/Tiltfile` carries them, commented, with the parts that are easy to get wrong explained next to them — building an image, a shared base built once, naming and ordering resources, reloading a gateway on a config change, kustomize flags, and reaching a service directly.
|
||||
43
rig/docs/notes/addons.md
Normal file
43
rig/docs/notes/addons.md
Normal file
@@ -0,0 +1,43 @@
|
||||
# ctrl/addons.sh and ctrl/addons/*.sh
|
||||
|
||||
## addons.sh
|
||||
|
||||
Each addon is its own idempotent script — adding one is adding a file, not editing a dispatcher. `ADDONS` names them, in install order; the overlay's `addons/<name>.sh` is found before rig's `ctrl/addons/<name>.sh`, and every one runs from rig's `ctrl/` with `RIG_CTRL` exported, wherever its file lives (see [overlay.md](overlay.md)).
|
||||
|
||||
## What rig ships, and what it does not
|
||||
|
||||
rig's own addons make the *cluster* work, and are useless outside one: metallb, cert-manager, metrics-server. Things a workload happens to need — a database, a cache, a scheduler — are the workload's, and which workload needs which is not rig's business, so they live with the overlay. `examples/data/addons/` has postgres, redis and airflow as a worked example; an overlay that wants them copies them in.
|
||||
|
||||
|
||||
## cert-manager.sh
|
||||
|
||||
In a regulated estate almost everything is TLS, so the interesting question
|
||||
during onboarding is "does this service present a cert my client trusts" — not
|
||||
"can I reach a public ACME server". A local CA answers that offline, which is
|
||||
also what makes the air-gapped profile usable.
|
||||
|
||||
## metallb.sh — why it matters
|
||||
|
||||
Real manifests use LoadBalancer, because a real cluster has one. On a bare kind
|
||||
cluster those Services sit at `EXTERNAL-IP <pending>` forever with no error
|
||||
anywhere — the deployment looks fine and simply is not reachable. Without MetalLB,
|
||||
every such Service has to be edited to NodePort, which means the local manifests
|
||||
stop matching the ones being modelled.
|
||||
|
||||
The address pool is derived from the kind Docker network at install time, not
|
||||
hardcoded: Docker picks that subnet, it differs between machines, and a pool
|
||||
outside it is silently unroutable.
|
||||
|
||||
## metallb.sh — waiting for the controller
|
||||
|
||||
`kubectl wait` on a selector errors out immediately when nothing matches yet, and
|
||||
right after apply the ReplicaSet has not created the pod — so it loses a race it
|
||||
looks like it should win. `rollout status` waits for the Deployment itself and
|
||||
handles the not-yet-created case.
|
||||
|
||||
## metrics-server.sh
|
||||
|
||||
kind nodes serve kubelet metrics over a self-signed cert, so the standard
|
||||
manifest never becomes ready without `--kubelet-insecure-tls`. That is fine here
|
||||
(it is a local cluster) and is the single most common reason metrics-server sits
|
||||
at 0/1 on kind.
|
||||
50
rig/docs/notes/check.md
Normal file
50
rig/docs/notes/check.md
Normal file
@@ -0,0 +1,50 @@
|
||||
# ctrl/check.sh
|
||||
|
||||
## Purpose
|
||||
|
||||
Readiness check: is this machine ready to run rig?
|
||||
|
||||
It reports and instructs; it never silently fixes anything. Everything it finds is either already fine, or something a human has to decide on.
|
||||
|
||||
Runs ctrl/deps.sh host detection in a container when Docker is the only thing installed, or directly when the toolchain is already present. Then adds the checks that need this repo's config: profile sanity, CA trust, port clashes.
|
||||
|
||||
## memory
|
||||
|
||||
A profile on a box that is already full is the most common first failure, and it presents as pods stuck Pending rather than anything that says "memory". The check warns; it never blocks. Whether to try anyway is the user's call.
|
||||
|
||||
## mb_of
|
||||
|
||||
MEMINFO and OVERCOMMIT_FILE exist only so the tight and does-not-fit branches can be exercised against another machine's real numbers; in normal use they are the kernel's own files.
|
||||
|
||||
## NODE_MB
|
||||
|
||||
NODE_MB (what one node costs) comes from load_config (lib/config.sh), where its measurement is recorded. It lives there, not here, because the memory tool and every standalone kit need the same number: a copy of it is how rigmini.sh came to say 2 GB per node long after rig had measured 800 MB.
|
||||
|
||||
## container_mb
|
||||
|
||||
Every running container's working set in MB, tagged with the kind cluster it belongs to ('-' when it is not kind). docker stats reports usage minus page cache, which is what actually competes: cache is handed back under pressure. Counting only kind would hide the usual culprit on a managed workspace, where the memory is held by other containers entirely.
|
||||
|
||||
## ours_mb / still_mb
|
||||
|
||||
Once this environment's own cluster is running, its real footprint is already out of MemAvailable and the per-node estimate stops being relevant. Subtracting the measurement from the estimate would count the same memory twice, and a running cluster that happens to sit under 800 MB would still "need" the gap.
|
||||
|
||||
## ports: our own cluster
|
||||
|
||||
A port held by THIS environment's own cluster is not a clash; it is the thing working. Reporting it as a problem every time the cluster is up would train people to ignore this section, which is the opposite of the point.
|
||||
|
||||
The ports are extracted with a second grep rather than `tr -d ':->'`: in tr, ':->' is the character RANGE ':' to '>', which does not contain '-', so the trailing dash survives and nothing ever matches.
|
||||
|
||||
## Compact by default
|
||||
|
||||
`make check` prints one line per question — host, toolchain, and for this rig: cluster, memory,
|
||||
ports, registry, addons — and adds detail only where something needs attention (`!` lines, the
|
||||
"held elsewhere" list when memory is tight, the clashing port). `make check all` prints every fact,
|
||||
as the full report did before 2026-09-17. `deps.sh detect all` is the same switch for the host part,
|
||||
so the standalone `rigdeps.sh detect` is short too. Changed because the long report buried the few
|
||||
lines that mattered.
|
||||
|
||||
## overlay and kind config
|
||||
|
||||
The rig block names the overlay when one is set (with `all`: what it provides — rig.env, manifests, kind config, addons, Tiltfile — and where the manifests and kind config resolved to).
|
||||
|
||||
Two `!` lines belong to overlays. An older ctrl/.env that still pins `MANIFESTS_DIR=ctrl/k8s/overlays/dev` — rig's examples, before they moved — is reported; load_config ignores it until then. And a kind config without the containerd `config_path` patch is reported whenever a registry mode needs it: registry.sh writes per-host config into certs.d, containerd only reads it if the cluster was created with that patch, and an overlay's own kind file replaces rig's whole template, so dropping it is easy and fails silently.
|
||||
15
rig/docs/notes/cluster.md
Normal file
15
rig/docs/notes/cluster.md
Normal file
@@ -0,0 +1,15 @@
|
||||
# ctrl/cluster.sh
|
||||
|
||||
## Why list and free live here
|
||||
|
||||
`list` and `free` live in `cluster.sh` rather than in a separate script because a
|
||||
near-identical second name (cluster / clusters) is a trap — you reach for one and
|
||||
get the other. One target, one file, unambiguous subcommands.
|
||||
|
||||
## Idempotent means convergent
|
||||
|
||||
"Idempotent" here means convergent, not "exits early if the cluster exists".
|
||||
That distinction matters: an interrupted first run can leave a cluster created
|
||||
but not finished, and returning early on the re-run would strand it there. The
|
||||
create step is conditional; every step after it always runs, and each one is
|
||||
individually idempotent.
|
||||
110
rig/docs/notes/config.md
Normal file
110
rig/docs/notes/config.md
Normal file
@@ -0,0 +1,110 @@
|
||||
# ctrl/lib/config.sh
|
||||
|
||||
## Purpose and precedence
|
||||
|
||||
The ecosystem convention is that scripts are standalone with no shared log library, and that still holds. This file is not a logging lib; it is the single definition of how the config layers compose, which every script has to agree on exactly. Precedence, weakest first:
|
||||
|
||||
```
|
||||
built-in defaults in load_config; fill only what nothing else set
|
||||
ctrl/versions.env pinned toolchain + image digests (committed)
|
||||
ctrl/env.d/<profile> how this machine reaches the world (OPTIONAL, examples ship as *.env.example)
|
||||
<overlay>/rig.env what runs: addons, namespaces, images (OPTIONAL, lives with the overlay)
|
||||
ctrl/.env machine-local values and secrets (gitignored)
|
||||
the caller's env `make cluster up PROFILE=<name>` (always wins)
|
||||
```
|
||||
|
||||
That last rule is why this is more than a few `source` lines: .env sets PROFILE, so without snapshotting it would silently override the PROFILE the user just typed on the command line.
|
||||
|
||||
Run from ctrl/.
|
||||
|
||||
## CONFIG_OVERRIDABLE
|
||||
|
||||
Values a user can reasonably override per-invocation. Anything set in the environment when load_config runs is restored after the files are read. NODES is deliberately NOT here: it is read back out of the kind config, so the file is the one place that decides it.
|
||||
|
||||
REGISTRY_PORT and MANIFESTS_DIR were missing here while ctrl/.env set them, so the caller's env silently LOST to the file for those two, breaking the one precedence rule the header states. Both are now listed. OVERLAY joined with overlays, for the same reason: it is chosen per machine or per call.
|
||||
|
||||
## default_cluster_name
|
||||
|
||||
The environment's folder name — the overlay's when one is named, else rig's own — reduced to something kind accepts as a cluster name (a DNS label: lowercase alphanumerics and dashes). Run from ctrl/, so rig's folder is the parent.
|
||||
|
||||
## _from_ctrl, _abs_from_ctrl
|
||||
|
||||
Paths in the config are relative to rig's folder (MANIFESTS_DIR, OVERLAY) or to ctrl/ (KIND_CONFIG), or absolute. Scripts run from ctrl/, so `_from_ctrl` turns a rig-relative path into one usable from there, and `_abs_from_ctrl` into an absolute one for consumers outside bash (ports.sh active, the kind config's hostPath entries).
|
||||
|
||||
## derive_port_base
|
||||
|
||||
Base of this environment's 10-port block. cksum is used rather than $RANDOM or bash hashing because it is POSIX and returns the same value on every machine, which is what makes the block reproducible instead of merely unique.
|
||||
|
||||
## load_config: RIG_PORTABLE
|
||||
|
||||
RIG_PORTABLE skips the machine-local layer. config_snapshot sets it, so a generated standalone kit never carries this machine's .env, which holds local values and, by its own description, secrets.
|
||||
|
||||
## load_config: profiles are optional
|
||||
|
||||
A profile is an optional overlay, never a prerequisite. rig assumes no configuration: with no profile named, or no env.d/ at all, it runs on the built-in defaults. What IS an error is naming a profile that does not exist, because a typo must not quietly fall back to something else.
|
||||
|
||||
## load_config: overlays
|
||||
|
||||
An overlay is one folder, outside rig's version control, that holds what runs ([overlay.md](overlay.md)). `OVERLAY` names it; a named overlay that does not exist is an error, like a named profile. With none named, rig's own `examples/starter` is used if it is present — it sets nothing, so a plain rig resolves as it did before overlays — and a rig copied without `examples/` still resolves, with no manifests.
|
||||
|
||||
Its `rig.env` is layered after the profile and before ctrl/.env. It may not set PROFILE or OVERLAY, which are chosen before it loads, and the paths it sets are relative to the overlay (load_config rewrites them as it loads the file), so an overlay can be moved without editing it.
|
||||
|
||||
## load_config: identity follows the folder
|
||||
|
||||
Identity follows the FOLDER — the overlay's when one is named, else rig's — so copying either somewhere else and renaming it yields a distinct environment with no further edits. Without this, two copies would share one cluster and `make cluster down` in either would destroy the other's. It is also what lets a project carry rig at `<project>/rig/` without every such project's cluster being called `rig`.
|
||||
|
||||
## load_config: host ports
|
||||
|
||||
Host ports are a single shared namespace, so unlike the cluster name they cannot just follow the directory; they have to be spread out. Anything already set (ctrl/.env, a profile, the command line) wins; only the gaps are filled. See [ports.md](ports.md) for the reasoning.
|
||||
|
||||
## load_config: MANIFESTS_DIR
|
||||
|
||||
Where the workload's manifests live, relative to rig's folder or absolute: the overlay's `k8s/overlays/dev` unless something names another. It is the seam that lets the real manifests be versioned away from the installer. `none` means rig applies none (the overlay's Tiltfile does). A named folder that does not exist is an error; the old default `ctrl/k8s/overlays/dev`, pinned by older .env files, is ignored while that folder does not exist and reported by `make check`.
|
||||
|
||||
## load_config: NODE_MB
|
||||
|
||||
What one node costs, measured rather than guessed. On 2026-09-11 a minimal control-plane node ran at 620 MiB idle and ~728 MiB with a small mock, plus 16 MiB for the local registry: ~745 MiB of working set. 800 rounds that up, and agrees with the 800 MB observed independently on a larger rig. Worker nodes carry no etcd or apiserver and are lighter, so for a multi-node shape this errs high. It is the cluster alone: whatever you deploy comes on top.
|
||||
|
||||
It is set here rather than in check.sh because the memory tool and every standalone kit need the same figure.
|
||||
|
||||
## render_kind_config
|
||||
|
||||
Renders the kind config to stdout. sed rather than envsubst: envsubst is gettext-base, absent from a minimal Debian, and Docker is meant to be the only prerequisite. The variable list is explicit so a template cannot quietly start depending on something the caller does not set.
|
||||
|
||||
hostPath entries are resolved by the HOST dockerd, so HOST_WORKDIR and OVERLAY_DIR must stay host paths even when this runs inside the installer container. `${OVERLAY_DIR}` renders to the overlay's absolute path, for mounting its folders into the nodes.
|
||||
|
||||
## What a standalone kit needs to know
|
||||
|
||||
The kit generator (ctrl/standalone.sh) asks these questions so that it never has to know how configuration is stored. Where profiles live, which files are layered and what is derived are config.sh's business and can change freely; the generator only calls these functions.
|
||||
|
||||
## config_profiles
|
||||
|
||||
Every configuration rig can be run as, one per line: each profile file, or, when there are none, `default`, the built-in configuration load_config uses when no profile is named. Never empty, because rig never needs a profile.
|
||||
|
||||
## config_snapshot
|
||||
|
||||
The resolved configuration, as `declare -p` lines: exactly what load_config leaves behind, minus the machine-local layer. A kit freezes this in place of load_config, so it carries rig's decisions and not this machine's secrets.
|
||||
|
||||
```
|
||||
config_snapshot <profile> that profile, as any machine would resolve it
|
||||
config_snapshot --current what THIS machine runs: every overridable key as
|
||||
resolved here, handed back in as if typed on the
|
||||
command line, over the same portable resolution.
|
||||
Values derived from those choices follow them;
|
||||
anything else the local layer set (credentials)
|
||||
is not carried. config_left_out names it.
|
||||
```
|
||||
|
||||
Found by difference, not by a list: whatever load_config sets today, it sets. A list here would be one more place to forget a variable.
|
||||
|
||||
## config_left_out
|
||||
|
||||
What an export of this machine's configuration does NOT carry, by name only: keys the machine-local layer sets that are not choices a caller may override. They are this machine's own (registry and mirror credentials, mostly), so the target has to be told to supply them. Values are never printed.
|
||||
|
||||
## config_freeze
|
||||
|
||||
A replacement for load_config with a resolution frozen in (a profile, or --current; see config_snapshot), printed as a function definition for a standalone kit to carry. The generator embeds whatever this prints and interprets none of it, so what "frozen" means stays rig's decision.
|
||||
|
||||
It keeps load_config's one stated rule: the caller's env wins for anything in CONFIG_OVERRIDABLE. A kit therefore behaves like rig (`OUT_BIN=... rigdeps.sh` still works) rather than like a copy with everything pinned.
|
||||
|
||||
What freezing does give up, knowingly: values DERIVED from an overridable one are fixed at generation. Override CLUSTER and the ports stay the ones derived for the original name. Re-deriving would mean carrying the layering itself, which is exactly what a kit exists not to need.
|
||||
162
rig/docs/notes/deps.md
Normal file
162
rig/docs/notes/deps.md
Normal file
@@ -0,0 +1,162 @@
|
||||
# ctrl/deps.sh
|
||||
|
||||
## Purpose and safety
|
||||
|
||||
Toolchain installer: detect the host, install a pinned toolchain onto it, then report what it could not do.
|
||||
|
||||
It never runs the cluster, never uses sudo or apt, and writes only into `$OUT_BIN` (default `~/.local/bin`). Everything that would touch the host proper — systemd, inotify limits, `.wslconfig`, docker group — is REPORTED for a human to decide on, never performed. That is what makes it safe to run on a machine that already has a working setup.
|
||||
|
||||
## Usage
|
||||
|
||||
Normally via `make deps`, or directly:
|
||||
|
||||
```
|
||||
deps.sh detect # report host facts only, change nothing
|
||||
deps.sh list # the pinned versions
|
||||
deps.sh verify [core|dev] # run what is installed and see if it works
|
||||
deps.sh fetch [core|dev] [--to DIR] # download + verify into DIR
|
||||
deps.sh install [core|dev] # detect, fetch, install, report
|
||||
```
|
||||
|
||||
Tiers: `core` is kubectl + jq (talk to a cluster); `dev` adds kind and tilt. Default is dev.
|
||||
|
||||
## Container vs bare host
|
||||
|
||||
Runs both inside the installer container and bare on a host. Inside the container, host files are read through `$HOST_ROOT` (mount `/` as `:ro`); bare, it falls back to `/`.
|
||||
|
||||
Host FILES (`/etc/...`, `/mnt/c/...`) must be read through the mount. Kernel-level facts (kernel version, meminfo, inotify) are shared with the container, so the container's own view is already the host's.
|
||||
|
||||
## INVOKED_FROM
|
||||
|
||||
Keep the caller's cwd so a relative `--to` resolves where the user expects, not against `ctrl/` once we've moved.
|
||||
|
||||
## load_config
|
||||
|
||||
Pins arrive through `load_config` like every other setting, not by sourcing `versions.env` here. That is what lets `make standalone` freeze them into a one-file installer: configuration has exactly one way in.
|
||||
|
||||
## mb_of
|
||||
|
||||
A `/proc/meminfo` field in MB, 0 if the field is absent. `MEMINFO` exists so the tight and does-not-fit branches can be exercised against a real machine's numbers from somewhere else; in normal use it is always `/proc/meminfo`.
|
||||
|
||||
## require_amd64
|
||||
|
||||
The pins are amd64. Rather than download something that cannot execute and let it fail as "cannot execute binary file: Exec format error", say so here and hand over the commands that produce the right checksums.
|
||||
|
||||
## pkg_install_cmd
|
||||
|
||||
This never runs a package manager. It names one so the reported action is something you can paste, on the distro you are actually on — an apt line on Amazon Linux 2 is a wrong answer dressed up as help.
|
||||
|
||||
## require_linux
|
||||
|
||||
Windows outside WSL — Git Bash, MSYS, Cygwin — looks close enough to work and then fails in a pile of confusing ways: no /proc, no docker socket, none of the tooling. Detectable, so name it instead.
|
||||
|
||||
## detect: memory
|
||||
|
||||
In MB. Whole gigabytes lose nearly half a GB on exactly the machines where it matters: 1874 MB available used to print as "1 GB". Facts only — whether that is enough depends on the profile, which `check.sh` knows and this does not.
|
||||
|
||||
## detect: overcommit
|
||||
|
||||
How the kernel answers an allocation it cannot really satisfy. With 1 it always says yes and settles up later with the OOM killer, so a cluster that starts cleanly can still lose processes afterwards.
|
||||
|
||||
## detect_wsl: systemd
|
||||
|
||||
systemd is off by default in WSL, and the ingress/DNS paths that use a host service need it. Enabling it requires a Windows-side restart, which cannot be issued from inside the distro.
|
||||
|
||||
## watch_hostile_fs
|
||||
|
||||
Not a path check: `/mnt` is an ordinary mount point and an ext4 disk mounted there is perfectly fine. What matters is the filesystem. The Windows drives arrive as 9p (WSL2) or drvfs (WSL1); network and fuse mounts behave the same way. None of them deliver inotify events, so anything watching files goes quiet without saying why.
|
||||
|
||||
## detect_libc
|
||||
|
||||
tilt is the one binary here that needs a recent glibc. MEASURED, not guessed: tilt 0.37.6 on Amazon Linux 2 (glibc 2.26) fails with
|
||||
|
||||
```
|
||||
/lib64/libc.so.6: version `GLIBC_2.34' not found (required by .../tilt)
|
||||
```
|
||||
|
||||
which names a symbol rather than the problem. Amazon Linux 2 is a stock WorkSpaces bundle, so this is the likely case, not an exotic one. Report the version now; `verify` catches the actual failure after installing.
|
||||
|
||||
## detect_prereqs
|
||||
|
||||
What this script needs to do its own job. Reported here so `detect` answers "will install work?" instead of leaving you to find out one download in. Amazon Linux 2 ships without tar, which is exactly the surprise this catches.
|
||||
|
||||
## detect_docker
|
||||
|
||||
Reachability of the daemon is the real question, and the CLI is only how we ask it. When this runs inside the installer container, Docker necessarily exists on the host — otherwise nothing would be executing — so a missing CLI in there is an installer packaging bug, not a host problem.
|
||||
|
||||
The kind-node count check must be an `if`, not `[ ] && echo`: as the last statement in the function the latter returns 1 when the count is zero, and `set -e` then kills the caller. That is the fresh-machine case — no clusters yet — so the bug only ever shows up where it does most harm.
|
||||
|
||||
## fetch_tgz: --no-same-owner
|
||||
|
||||
Extracting as root would otherwise restore the uid/gid baked into the archive (some ship as uid 1001), leaving a binary the host user does not own.
|
||||
|
||||
## fix_ownership
|
||||
|
||||
The installer runs as root so it can reach the docker socket, which means everything it writes into a mounted volume lands root-owned and unusable from the host. Hand it back to whoever owns the mount point (the host user created that directory before mounting it).
|
||||
|
||||
kind writes the kubeconfig as root too; `fetch` hands that back as well when it's a mounted host directory rather than container-local state.
|
||||
|
||||
## Tiers (CORE_TOOLS, DEV_TOOLS)
|
||||
|
||||
Two tiers, because not every machine should get cluster tooling.
|
||||
|
||||
- `core` — kubectl, jq: talk to a cluster someone else runs. Nothing that creates one. Appropriate on a managed or corporate-issued machine where development tools are not wanted by default.
|
||||
- `dev` — core plus kind and tilt: build clusters and hot-reload into them.
|
||||
|
||||
The split exists because "install the toolchain" is not one decision: on a managed workspace the right answer is kubectl and nothing else.
|
||||
|
||||
No helm: every addon installs with `kubectl apply -f <url>`, so nothing here has ever invoked it. Add it back the day something actually needs a chart.
|
||||
|
||||
ctlptl is `dev` rather than `core` for the same reason kind is: core is "talk to a cluster someone else runs", and ctlptl builds them. It earns its place because it is what wires a cluster to a local registry — without one, an unqualified image name resolves to `docker.io/library/<name>` and there is nothing structural stopping a push there.
|
||||
|
||||
docker-compose is `dev` for the same reason, and is here because the distro docker packages ship the daemon and CLI but frequently not the compose plugin — so `docker compose up` fails with "unknown command" on an otherwise working Docker, and nothing about that message names the missing piece.
|
||||
|
||||
## What is already on this machine (pin_of)
|
||||
|
||||
A tool already on PATH at its pinned version is left where it is. Without this, install downloads a second copy into `OUT_BIN` and then reports the first one as shadowed — noise, and wrong, when both are the same version. That is the normal state of any machine someone set up by hand, whatever directory they happened to choose.
|
||||
|
||||
## reported_version
|
||||
|
||||
Each tool spells the version question differently, and kubectl has to be told `--client` or it goes looking for a server to ask.
|
||||
|
||||
## version_matches
|
||||
|
||||
Matched as a whole version token, so 0.37.6 never matches 10.37.60, with the leading v optional either side: kind says v0.32.0, jq says jq-1.8.2, and tilt says v0.37.6 against a pin of 0.37.6.
|
||||
|
||||
Bash's own regex rather than grep, deliberately. grep is not the same program on every machine — some builds reject patterns that others accept — and a failed grep inside a count reads exactly like a zero.
|
||||
|
||||
## want / DEPS_ONLY
|
||||
|
||||
`DEPS_ONLY` narrows a fetch to the tools it names. Unset means the whole tier, which is what an explicit `deps.sh fetch` always gets: "download these into DIR" must not quietly skip something because this machine happens to have it. Only `install()` sets it, to what `detect_toolchain` found missing or mismatched.
|
||||
|
||||
## detect_toolchain: compose
|
||||
|
||||
compose is the one tool that is normally NOT a binary on PATH. It is a docker CLI plugin, so a machine where `docker compose` works perfectly has no `docker-compose` to find — and probing only PATH would report it missing and re-download a copy that is already there. That is the exact noise the version-aware skip exists to prevent, so ask docker instead.
|
||||
|
||||
## verify_tools
|
||||
|
||||
Installing into a directory that sits early in PATH silently replaces whatever the machine was already using — which on a shared or client machine can break unrelated work (kubectl more than one minor away from a cluster is the common one). Say so; never decide it for them.
|
||||
|
||||
Downloading a verified binary proves it is the right file, not that this machine can run it. On an old distro tilt fails here, with a linker error about a missing symbol, and finding that out now beats finding out during a first cluster build.
|
||||
|
||||
Output is not piped into `head`. With `pipefail` set, a tool that prints more than one line gets SIGPIPE when head closes the pipe, and the pipeline reports 141 — so a working kubectl was announced as "does not run here", with its own correct version string as the evidence. The first line is taken afterwards, from the string.
|
||||
|
||||
## install_compose_plugin
|
||||
|
||||
A copy in `OUT_BIN` only gives you `docker-compose`. That hyphenated form is the retired v1 spelling; every compose file written in the last few years assumes `docker compose`, which resolves plugins BY NAME out of a plugin directory. So the binary is fetched like any other and then linked, in your own home — no root, and nothing outside it.
|
||||
|
||||
If something else already owns that name — docker-desktop and some distro packages install a real file there — overwriting it would take the plugin away from whatever put it there, so say so and let the user decide.
|
||||
|
||||
## install
|
||||
|
||||
The plugin is linked only when compose was one of the things fetched: linking a binary that is already satisfied elsewhere on PATH would point the plugin at a copy rig did not install.
|
||||
|
||||
The "put OUT_BIN on PATH" advice is only worth giving when something actually landed in `OUT_BIN`. When every tool was satisfied elsewhere, `OUT_BIN` may reasonably be off PATH, and telling the user to add it would be advice to fix nothing.
|
||||
|
||||
## main: argument shift
|
||||
|
||||
Read the command, THEN shift — and shift only if there is something there. A bare `shift` with no positional parameters returns 1, and under `set -e` that ended the script before a single line was printed: running this with no arguments at all, the documented default, did nothing and said nothing.
|
||||
|
||||
## manifest / manifests
|
||||
|
||||
The manifests rig's own addons apply are pinned in versions.env like the binaries, and fetched by the same code: `resolve_url` for the source (upstream, artifactory, baked), `verify` for the sum. `manifest <NAME>` makes one present in `vendor/manifests/` (rig's folder, gitignored) and prints only its path, so an addon can apply it; a cached copy whose sum still matches is reused, one that does not is fetched again. `manifests [--to DIR]` fetches all three, which is how the deps-full image bakes them and how an offline machine is given them. Why the addons stopped applying URLs is in versions.md.
|
||||
14
rig/docs/notes/docs.md
Normal file
14
rig/docs/notes/docs.md
Normal file
@@ -0,0 +1,14 @@
|
||||
# ctrl/docs.sh
|
||||
|
||||
## Serving without the cluster or python
|
||||
|
||||
The docs are the instructions for building the cluster, so they must work before
|
||||
anything else exists. That rules out serving them from the cluster, and it rules
|
||||
out `python -m http.server` too — a minimal Debian has no python3. What it does
|
||||
have, by definition, is Docker: the single prerequisite rig already demands. So a
|
||||
throwaway nginx container serves a read-only bind mount.
|
||||
|
||||
## Committed SVGs
|
||||
|
||||
Rendered SVGs are committed alongside their `.dot` sources for the same reason:
|
||||
the pages have to read on a machine with no Graphviz installed.
|
||||
64
rig/docs/notes/env.md
Normal file
64
rig/docs/notes/env.md
Normal file
@@ -0,0 +1,64 @@
|
||||
# ctrl/.env.example, ctrl/env.d/*.env.example
|
||||
|
||||
## ctrl/.env.example: header
|
||||
|
||||
Machine-local config. Copy to ctrl/.env (gitignored) and edit. The cluster SHAPE is an optional profile in ctrl/env.d/ — see the *.env.example there. The architecture MODEL lives in arch/<name>.json — not in .env either.
|
||||
|
||||
## ctrl/.env.example: CLUSTER
|
||||
|
||||
The kubectl context becomes kind-<CLUSTER>. LEAVE THIS UNSET unless you need a name that differs from the directory — it defaults to this folder's name, which is what makes the folder copyable: copy it, rename it, and you get a separate environment with no edits.
|
||||
|
||||
## ctrl/.env.example: host ports
|
||||
|
||||
LEAVE UNSET — they derive from the directory name so several environments coexist without negotiating (see ctrl/ports.sh). `make check` shows this environment's block; `bash ctrl/ports.sh persist` writes it into ctrl/.env so it stops being derived and becomes fixed. Set a value only to override.
|
||||
|
||||
## ctrl/.env.example: OVERLAY
|
||||
|
||||
The folder that holds what runs — its settings (`rig.env`), manifests, addons, Tiltfile — kept outside rig's version control: `local/<name>` (gitignored), or a repo of its own anywhere. Relative to rig's folder, or absolute. Unset, rig runs its own `examples/starter`. The cluster, context and port block follow the overlay's folder name. See [overlay.md](overlay.md).
|
||||
|
||||
## ctrl/.env.example: MANIFESTS_DIR
|
||||
|
||||
Where the manifests live. Leave it unset: the overlay's `k8s/overlays/dev` is the default. Set it only to point somewhere else, relative to rig's folder or absolute:
|
||||
|
||||
MANIFESTS_DIR=../platform-manifests/overlays/dev
|
||||
|
||||
Older copies of this file set `MANIFESTS_DIR=ctrl/k8s/overlays/dev`, rig's examples before they moved to `examples/`. That value is ignored while the folder does not exist, and `make check` says to delete the line.
|
||||
|
||||
## ctrl/.env.example: DEPS_SOURCE
|
||||
|
||||
Where the installer fetches the pinned binaries from.
|
||||
|
||||
- `upstream` — GitHub releases / dl.k8s.io (needs internet)
|
||||
- `artifactory` — a generic repo; what a locked-down client usually allows
|
||||
- `baked` — already inside the installer image; no network at all
|
||||
|
||||
## ctrl/.env.example: registry secrets
|
||||
|
||||
The registry mode comes from the profile (REGISTRY_MODE). REGISTRY_REMOTE_URL, REGISTRY_USER and REGISTRY_PASSWORD are the secrets it needs, required for mirror/remote.
|
||||
|
||||
## ctrl/.env.example: REGISTRY_CA_FILE
|
||||
|
||||
Corporate root CA, if Artifactory is fronted by an internal CA (it usually is). Trust has to reach THREE places and nothing does it for you: the host docker daemon, every kind node's containerd, and any in-cluster client. registry.sh handles the first two; check.sh reports when it's configured but not trusted. Symptom when missing: `x509: certificate signed by unknown authority`.
|
||||
|
||||
## env.d/*.env.example: profiles in general
|
||||
|
||||
EXAMPLE PROFILES. rig needs none of these: with no profile it runs on its built-in defaults (lib/config.sh). To use one, copy it to <name>.env in ctrl/env.d/ and name it — PROFILE=<name> in ctrl/.env, or on the command line. It then overlays the defaults; anything it does not set, they still supply. An activated `<name>.env` is gitignored: it is this machine's choice.
|
||||
|
||||
A profile says how this machine reaches the world — a registry mirror, an air-gapped install. What runs is an overlay's business ([overlay.md](overlay.md)); its `rig.env` layers above the profile.
|
||||
|
||||
## env.d/mirror.env.example
|
||||
|
||||
mirror — images through a pull-through cache of an internal registry, with TLS and metrics addons. More nodes or port mappings: edit the kind config (rig's, or the overlay's).
|
||||
|
||||
### Real ports (80/443)
|
||||
|
||||
Ports derive from the directory name by default (see ctrl/ports.sh), so several environments run side by side.
|
||||
|
||||
Opt in to the real ports only when this is the ONLY environment and nothing else owns :80. They fail to bind otherwise, and docker reports it as an opaque "failed to bind host port 0.0.0.0:80/tcp: address already in use" halfway through cluster creation. `make check` checks before you spend the time. Uncommenting also means only one environment can exist at a time.
|
||||
|
||||
|
||||
## env.d/offline.env.example
|
||||
|
||||
offline — air-gapped. Everything comes from a local registry that was loaded ahead of time; nothing reaches the internet. Pair with the deps-full image (DEPS_SOURCE=baked) so the toolchain install is offline too, and so the manifests metallb and the other addons apply come from the image, verified, rather than from GitHub (see Dockerfile.deps.md). Their container images still have to be preloaded.
|
||||
|
||||
The heavier addons are left out to keep first boot viable.
|
||||
29
rig/docs/notes/kind-config.md
Normal file
29
rig/docs/notes/kind-config.md
Normal file
@@ -0,0 +1,29 @@
|
||||
# ctrl/k8s/kind-config.yaml.tpl
|
||||
|
||||
## Why a template
|
||||
|
||||
The cluster: one node by default — add nodes or port mappings by editing the file, then `make cluster reset`.
|
||||
|
||||
It is a TEMPLATE rather than a plain kind-config.yaml because a rig (or an overlay) is copied and renamed to make a second environment, and both the cluster name and the host port follow the folder. A checked-in literal would make every copy collide on both — which is exactly why every other project here, with its literal kind-config.yaml, has only one of itself. ctrl/cluster.sh renders it with sed — not envsubst, which is gettext-base and absent from a minimal Debian, and rig's whole premise is that Docker is the only prerequisite.
|
||||
|
||||
A kind config is fixed at creation: to change the cluster, edit the file, then `make cluster reset`. lib/config.sh reads the node count back out of it, so nothing restates it.
|
||||
|
||||
## An overlay's own kind config
|
||||
|
||||
An overlay may carry its own `kind-config.yaml.tpl` (see [overlay.md](overlay.md)); it replaces this whole file, rendered the same way, so start from a copy of this one. Keep the containerd `config_path` patch: `make check` reports its absence whenever a registry mode needs it. A project that builds its own cluster through rig can also pass any file as `KIND_CONFIG=<path>`.
|
||||
|
||||
## Substituted variables
|
||||
|
||||
Substituted by ctrl/cluster.sh: CLUSTER, NODE_IMAGE, HTTP_PORT, HOST_WORKDIR (rig's folder), OVERLAY_DIR (the overlay's folder, for mounts). The header comment names them without the `${...}` braces so that line survives the substitution.
|
||||
|
||||
## Node count
|
||||
|
||||
The node count is READ BACK from this file by lib/config.sh, so this YAML is the source of truth for it — there is no second place to update.
|
||||
|
||||
## containerdConfigPatches
|
||||
|
||||
Point containerd at a certs.d directory. registry.sh drops per-host hosts.toml files in there afterwards, so switching registry mode never requires recreating the cluster.
|
||||
|
||||
## extraPortMappings
|
||||
|
||||
One NodePort bridged to the host; an in-cluster gateway owns it. There is deliberately no ingress controller — they pin a narrow window of k8s versions, and running a trailing-edge control plane is the point.
|
||||
96
rig/docs/notes/mem.md
Normal file
96
rig/docs/notes/mem.md
Normal file
@@ -0,0 +1,96 @@
|
||||
# ctrl/mem.sh
|
||||
|
||||
## Purpose
|
||||
|
||||
How much memory this machine will actually give you before something dies. This is rig's memory tool, and the standalone rigmini.sh is generated from this file.
|
||||
|
||||
There are two numbers and they are rarely the same. `status` reports what the machine ADVERTISES and what is quietly capping it. `push` finds what it will SURVIVE, by allocating until it stops. `all` does both and weighs the result against what this profile's cluster needs.
|
||||
|
||||
The gap between them is the whole reason this exists. Under WSL the cap lives in .wslconfig; in a container or a managed workspace it is a cgroup limit, and there /proc/meminfo reports the HOST's memory while the kernel kills you at a fraction of it. A script that only read MemTotal would confidently report 32 GB on a box that OOMs at 2.
|
||||
|
||||
Runs on native Linux and under WSL. On WSL the memory you see is a VM allocation that can be raised, and the commonest failure is raising it without restarting, so status compares what .wslconfig says with what actually booted.
|
||||
|
||||
It reports and instructs. It never raises a limit, frees anything or installs a package. The one write it can make is `backup`, which copies .wslconfig beside itself, so that `restore` has something to put back after a hand edit.
|
||||
|
||||
Usage:
|
||||
|
||||
```
|
||||
mem.sh status what it has, what caps it
|
||||
mem.sh push [--to GB] [--to-oom] climb until it stops
|
||||
mem.sh all [--budget GB] both, then the verdict
|
||||
mem.sh backup | restore .wslconfig, WSL only
|
||||
```
|
||||
|
||||
## require_linux
|
||||
|
||||
Windows outside WSL (Git Bash, MSYS, Cygwin) looks close enough to work and then fails in a pile of confusing ways: no /proc, no docker socket, none of the tooling. It is detectable, so name it instead.
|
||||
|
||||
## CG_MAX_FILE / CG_CUR_FILE
|
||||
|
||||
Where a cgroup records this cgroup's own limit and usage. Set once by find_cgroup, because every later reading needs both, and hunting for the files on each call would be the slow part of the poll loop.
|
||||
|
||||
## find_cgroup
|
||||
|
||||
Inside a container the cgroup namespace makes the top of the tree BE the container's own cgroup, so the unqualified path is already the right one. On a host it is the root cgroup, which is never limited; hence the second attempt via /proc/self/cgroup, which names the slice this shell is in.
|
||||
|
||||
## cgroup_cap_mb
|
||||
|
||||
Returns the cap in MB, or "" when there is none worth reporting. cgroup v2 spells unlimited "max"; v1 spells it as a number near 2^63, which is why this compares against MemTotal rather than testing for a magic value. A "limit" above the machine's own memory is not a limit, however it is written.
|
||||
|
||||
## headroom_mb
|
||||
|
||||
How much room is left RIGHT NOW, from whichever accounting actually governs. In a capped container /proc/meminfo describes the host and is worse than useless for this: it would report tens of gigabytes free on a box that is one allocation from being killed.
|
||||
|
||||
## wslconfig_path
|
||||
|
||||
/mnt/c/Users can hold several real accounts (a renamed login leaves the old directory behind), so picking the first alphabetically is a coin toss. Ask Windows, then fall back to whichever profile actually owns a config.
|
||||
|
||||
## status: overcommit
|
||||
|
||||
overcommit_memory=0 is the default heuristic: a large allocation is granted on a guess, and the reckoning arrives later as an OOM kill rather than as a failed malloc. It is why `push` touches every page it asks for.
|
||||
|
||||
## status: WSL
|
||||
|
||||
WSL keeps its cap on the Windows side, in a file this shell can read but not usefully apply: the change costs a full VM restart. Report it, and report the commonest mistake, which is editing it and not restarting.
|
||||
|
||||
## backup
|
||||
|
||||
Backups are timestamped and never overwritten: a backup that can destroy itself on a second run is not a backup.
|
||||
|
||||
## restore
|
||||
|
||||
Newest is the right default (undo the last edit), but if you backed up *after* editing, the state you want is older. The rest are shown so a no-op restore is obviously a no-op rather than a mystery.
|
||||
|
||||
## allocator
|
||||
|
||||
The child allocates and stops itself; the parent only watches. That split is the point: under --to-oom the allocating process is expected to be killed, and something has to survive to say how far it got.
|
||||
|
||||
### OOM score
|
||||
|
||||
The child raises its own OOM score to the maximum so the kernel picks THIS process first. Raising needs no privilege (only lowering does). Without it, the kernel is free to choose your shell, your ssh session or dockerd; on a box you are still using, that is not an acceptable coin toss.
|
||||
|
||||
### Writing straight into the array element
|
||||
|
||||
Each chunk is written STRAIGHT INTO the array element (`printf -v "arr[$i]"`). The obvious spelling, building one chunk and `arr+=("$chunk")`, costs three copies per step, not one: the template stays resident, expanding "$chunk" makes a temporary word, and the append makes the element. A 128 MB step then needs 384 MB transiently, and on a small box it is killed on the first append while reporting a third of the true ceiling.
|
||||
|
||||
printf -v into a subscript also means every page is written, so it is resident rather than merely promised: the only kind of allocation that measures anything under heuristic overcommit.
|
||||
|
||||
### First swap
|
||||
|
||||
Worth calling out separately from the ceiling: this is where the box stops being fast and starts being unusable, which for a scheduler is a different and earlier problem than being killed.
|
||||
|
||||
## push: step size
|
||||
|
||||
A step is worth about a sixty-fourth of the ceiling: enough resolution to find the edge, few enough lines to read, and small enough that the transient cost of one allocation never dominates a small box. A fixed size cannot do all three: 128 MB is fine on 16 GB and absurd on 512 MB.
|
||||
|
||||
## push: floor
|
||||
|
||||
Stop with a cushion rather than riding it to the kill. How big a cushion depends on what it is protecting. Under a cgroup cap, running out kills only this container's own processes, so it need cover no more than the shell that prints the result, and a 512 MB cushion on a 1 GB box would halve the answer. On a host there is everything else to protect, and the OOM killer does not promise to pick the process that caused the problem.
|
||||
|
||||
## push: Ctrl-C
|
||||
|
||||
INT kills the child and lets the summary print anyway, so an impatient Ctrl-C still tells you how far it got and, more importantly, still gives the memory back.
|
||||
|
||||
## push: claimed vs. measured
|
||||
|
||||
The gap between the claim and the measurement is the finding, but only when the BOX chose where to stop. An empty $stop means the child was ended rather than deciding to end; anything else (--to, the floor) is a stop we asked for, and flagging those as short of the ceiling would put a warning on every deliberately small run.
|
||||
171
rig/docs/notes/overlay.md
Normal file
171
rig/docs/notes/overlay.md
Normal file
@@ -0,0 +1,171 @@
|
||||
# Overlays: what runs lives outside rig
|
||||
|
||||
## Why
|
||||
|
||||
rig is the machine: the toolchain, the cluster, the registry, the port block, the
|
||||
dev loop's plumbing. What runs on it — the services, their manifests, their
|
||||
images, their settings — belongs to whoever owns that work, changes at a
|
||||
different rate, and often cannot be shared at all. Keeping both in one tree meant
|
||||
editing rig's own files to use it, and then carrying those edits into every copy.
|
||||
|
||||
So the use case is one folder, the **overlay**, kept outside rig's version
|
||||
control. rig reads it; rig never writes into it and never knows what is in it.
|
||||
The dependency points one way: an overlay knows about rig, rig knows about
|
||||
overlays in general and about none in particular.
|
||||
|
||||
## Where an overlay lives
|
||||
|
||||
```
|
||||
rig/local/<name>/ gitignored by rig: an overlay with no version control of its own,
|
||||
or a clone of its own repo
|
||||
anywhere/<name>/ a repo of its own, named by path
|
||||
<project>/ a project folder that carries rig at <project>/rig/ (the vendored
|
||||
layout, below)
|
||||
```
|
||||
|
||||
Name it with `OVERLAY` — in `ctrl/.env` for this machine, or per call:
|
||||
|
||||
```bash
|
||||
OVERLAY=local/myenv make cluster up
|
||||
```
|
||||
|
||||
Relative paths are relative to rig's folder. With none named, rig uses its own
|
||||
`examples/starter`, which sets nothing, so a plain rig behaves as it always did.
|
||||
An overlay that is named and missing is an error; it never falls back.
|
||||
|
||||
## What rig reads from it
|
||||
|
||||
Every piece is optional.
|
||||
|
||||
| in the overlay | what rig does with it |
|
||||
| --- | --- |
|
||||
| `rig.env` | a config layer (below). Any key a profile could set. |
|
||||
| `k8s/overlays/dev/` | the default `MANIFESTS_DIR`: rig's Tiltfile applies it with kustomize |
|
||||
| `kind-config.yaml.tpl` | the default `KIND_CONFIG`: the cluster's shape, rendered like rig's own |
|
||||
| `addons/<name>.sh` | an addon, found before rig's `ctrl/addons/<name>.sh` of the same name |
|
||||
| `Tiltfile` | the workload's half of the dev loop, included by rig's `ctrl/Tiltfile` |
|
||||
|
||||
Anything else in the folder is the overlay's own business: Dockerfiles, DAGs,
|
||||
folders of repos or data it mounts, its `.gitignore`, the secrets its kustomize
|
||||
generators read. rig does not look.
|
||||
|
||||
## Layers
|
||||
|
||||
```
|
||||
built-in defaults < ctrl/versions.env < ctrl/env.d/<profile>.env < <overlay>/rig.env < ctrl/.env < the caller
|
||||
```
|
||||
|
||||
A profile says how this machine reaches the world (a registry mirror, an
|
||||
air-gapped install); an overlay says what runs. `ctrl/.env` is still this
|
||||
machine's, and the caller still wins over everything.
|
||||
|
||||
`rig.env` may not set `PROFILE` or `OVERLAY`: both are chosen before it loads.
|
||||
The paths it sets (`MANIFESTS_DIR`, `KIND_CONFIG`) are relative to the overlay.
|
||||
`MANIFESTS_DIR=none` means rig applies no manifests and the overlay's Tiltfile
|
||||
does, e.g. when kustomize needs flags.
|
||||
|
||||
## Identity
|
||||
|
||||
With an overlay named, the cluster, the kubectl context and the port block
|
||||
follow the overlay folder's name, sanitised the same way a rig folder's name is.
|
||||
One rig can therefore serve several overlays, each in its own cluster, and a
|
||||
project that carries rig at `./rig` does not name every cluster `rig`.
|
||||
|
||||
`ports.sh persist` refuses while an overlay is set: it writes to rig's
|
||||
`ctrl/.env`, and a pin there would follow every overlay.
|
||||
|
||||
## The Tiltfile handoff
|
||||
|
||||
rig's `ctrl/Tiltfile` does rig's part — the context guard, `default_registry`,
|
||||
the manifests, the namespaces they use — then publishes the facts and includes
|
||||
the overlay's `Tiltfile`:
|
||||
|
||||
```
|
||||
RIG_CLUSTER RIG_CONTEXT RIG_HTTP_PORT RIG_HTTPS_PORT RIG_TILT_PORT RIG_REGISTRY RIG_OVERLAY_DIR
|
||||
```
|
||||
|
||||
Read them with `os.getenv`. An included Tiltfile runs from its own folder, so
|
||||
every relative path in it (`docker_build` contexts, `sync`, `deps`, `local`) is
|
||||
relative to the overlay — it never needs a path back into rig.
|
||||
|
||||
## Addons
|
||||
|
||||
An addon is a bash script run by `ctrl/addons.sh` from rig's `ctrl/`, with
|
||||
`RIG_CTRL` exported. It starts like this, sources the config and does its work:
|
||||
|
||||
```bash
|
||||
cd "${RIG_CTRL:?run it through rig: bash ctrl/addons.sh install}"
|
||||
source ./lib/config.sh
|
||||
load_config
|
||||
```
|
||||
|
||||
`OVERLAY_DIR` is set, so an addon can find files beside it
|
||||
(`$(_from_ctrl "$OVERLAY_DIR")/...`). rig's own `ctrl/addons/` holds only what
|
||||
makes a cluster work (metallb, cert-manager, metrics-server);
|
||||
`examples/data/addons/` shows workload ones.
|
||||
|
||||
## The kind config
|
||||
|
||||
An overlay's `kind-config.yaml.tpl` replaces rig's whole file, so start from a
|
||||
copy of `ctrl/k8s/kind-config.yaml.tpl` and keep its containerd `config_path`
|
||||
patch: `registry.sh` needs it, and `make check` says so when it is missing.
|
||||
`${OVERLAY_DIR}` renders to the overlay's absolute path, for mounts:
|
||||
|
||||
```yaml
|
||||
extraMounts:
|
||||
- hostPath: ${OVERLAY_DIR}/datadir
|
||||
containerPath: /rig/datadir
|
||||
```
|
||||
|
||||
A kind config is fixed when the cluster is created: after changing it,
|
||||
`make cluster reset`.
|
||||
|
||||
## The vendored layout
|
||||
|
||||
A project folder can carry rig inside it and be the overlay itself:
|
||||
|
||||
```
|
||||
<project>/
|
||||
Makefile the forwarder below
|
||||
rig.env k8s/ Tiltfile kind-config.yaml.tpl addons/ ...
|
||||
rig/ rig, placed as it is; tracked or ignored by the project, its call
|
||||
```
|
||||
|
||||
The forwarder runs rig with `OVERLAY` set to this folder. It passes `OVERLAY` in
|
||||
the environment, not as a make argument, so rig's own `$(shell ...)` sees it
|
||||
under make 4.3 as well:
|
||||
|
||||
```make
|
||||
HERE := $(patsubst %/,%,$(dir $(abspath $(lastword $(MAKEFILE_LIST)))))
|
||||
ifeq ($(wildcard $(HERE)/rig/Makefile),)
|
||||
$(error rig/ is missing — this folder is an overlay; put rig in ./rig)
|
||||
endif
|
||||
GOALS := $(or $(MAKECMDGOALS),help)
|
||||
.PHONY: $(GOALS)
|
||||
$(firstword $(GOALS)):
|
||||
@OVERLAY='$(HERE)' $(MAKE) --no-print-directory -C '$(HERE)/rig' $(GOALS)
|
||||
$(wordlist 2,$(words $(GOALS)),$(GOALS)):
|
||||
@:
|
||||
```
|
||||
|
||||
The cluster is then named after `<project>`, exactly as a copied rig named
|
||||
`<project>` was, so moving a copied rig to this layout keeps its cluster and ports.
|
||||
|
||||
`rig.env` holds no secrets, by this contract. A repository whose `.gitignore` has a
|
||||
broad `*.env` (a common secrets rule) would still hide it, so an overlay living in
|
||||
such a repo re-includes it in its own `.gitignore`: `!rig.env`.
|
||||
|
||||
## Moving a copied rig to an overlay
|
||||
|
||||
A rig copied into a project and edited there splits cleanly:
|
||||
|
||||
| was, in the copy | goes to |
|
||||
| --- | --- |
|
||||
| `ctrl/k8s/base`, `ctrl/k8s/overlays` | `k8s/` |
|
||||
| the workload parts of `ctrl/Tiltfile` | `Tiltfile` (paths now relative to the overlay) |
|
||||
| `ctrl/env.d/<name>.env` | `rig.env` |
|
||||
| edits to `ctrl/k8s/kind-config.yaml.tpl` | `kind-config.yaml.tpl` (`${HOST_WORKDIR}` → `${OVERLAY_DIR}`) |
|
||||
| workload addons | `addons/` |
|
||||
| Dockerfiles for the workload | beside the Tiltfile |
|
||||
| `ctrl/.env` | `rig/ctrl/.env` (this machine's; drop a `MANIFESTS_DIR=ctrl/k8s/overlays/dev` line) |
|
||||
| everything else of rig's | replaced by rig as it is |
|
||||
38
rig/docs/notes/ports.md
Normal file
38
rig/docs/notes/ports.md
Normal file
@@ -0,0 +1,38 @@
|
||||
# ctrl/ports.sh
|
||||
|
||||
## Why each environment gets a port block
|
||||
|
||||
New versions of a system mean new clusters on ONE machine, not new machines. Cluster name, kubectl context, registry container and image tag already derive from the directory name, so two copies never collide there, but host ports are a single shared namespace and would.
|
||||
|
||||
The block is derived from the directory name: stateless, stable, and requiring no coordination between copies that know nothing about each other.
|
||||
|
||||
```
|
||||
base = 20000 + (hash(slug) % 200) * 10
|
||||
+0 HTTP +1 HTTPS +2 TILT +3 REGISTRY (+4..9 reserved)
|
||||
```
|
||||
|
||||
20000+ deliberately avoids the ports something is already likely to hold: 80, 443, 3000, 5432, 8000, 8080.
|
||||
|
||||
Derivation is a default, not a decision. On first use the resolved block is written into ctrl/.env, so it becomes pinned, visible and editable rather than a number that appears from nowhere. Anything already in ctrl/.env wins.
|
||||
|
||||
## active
|
||||
|
||||
The resolved facts a consumer outside bash needs, machine-readable:
|
||||
|
||||
```
|
||||
CLUSTER KUBECONTEXT HTTP HTTPS TILT REGISTRY MANIFESTS_DIR OVERLAY_DIR
|
||||
```
|
||||
|
||||
Identity and ports together, because they are one fact set: both derive from a folder name (the overlay's when one is named) so that copies never collide. A consumer needs all of them or none, and fetching them separately is how two end up disagreeing. MANIFESTS_DIR and OVERLAY_DIR ride along because the one consumer that needs the addressing is the one that needs to know what to deploy and whose Tiltfile to include.
|
||||
|
||||
The two paths are absolute, or `-` when there is none: an empty field would shift every later one. OVERLAY_DIR was appended rather than inserted, so readers that take fields by position kept their indexes.
|
||||
|
||||
Space-separated, so the paths must not contain whitespace; `active` refuses rather than print a line that splits wrong. Everything else in rig already assumes that of paths; kind, docker and kubectl all do.
|
||||
|
||||
## persist, with an overlay
|
||||
|
||||
`persist` writes into ctrl/.env, which belongs to this rig, not to an overlay. With OVERLAY set, a block pinned there would follow every overlay this rig later runs, and two of them would then share ports — the collision the derivation exists to prevent. So it refuses and says so; an overlay's ports stay derived from its folder name.
|
||||
|
||||
`derive` answers a DIFFERENT question (what the directory name alone implies) and deliberately ignores ctrl/.env. Configuring anything from it would silently contradict the rule that "anything already in ctrl/.env wins". `active` is what anything downstream should read.
|
||||
|
||||
Why this exists at all: the cluster name is not the bare directory name. default_cluster_name() lowercases it and replaces every character outside [a-z0-9-], because it has to be a DNS label. Re-deriving that in another language is how a copy in `My_Project/` ends up guarding the wrong context.
|
||||
41
rig/docs/notes/registry.md
Normal file
41
rig/docs/notes/registry.md
Normal file
@@ -0,0 +1,41 @@
|
||||
# ctrl/registry.sh
|
||||
|
||||
## Registry modes
|
||||
|
||||
Registry plumbing. This is the seam — not a tool. Four modes, selected by
|
||||
`REGISTRY_MODE` in the active profile:
|
||||
|
||||
- **none** — Tilt builds straight into the node. No registry at all, and so no
|
||||
guard against an outward push: an unqualified image name means
|
||||
`docker.io/library/<name>`, and only Tilt's kind detection stands between that
|
||||
and a real push. Throwaway use only; every profile here now defaults to `local`
|
||||
instead.
|
||||
- **local** — a `registry:2` container wired into the cluster.
|
||||
- **mirror** — the same container, but configured as a pull-through cache of the
|
||||
corporate registry. This is what a locked-down client actually looks like:
|
||||
images originate from corp, you don't hammer it, and you keep working when the
|
||||
VPN drops.
|
||||
- **remote** — no local container; pull straight from the corporate registry
|
||||
using an imagePullSecret.
|
||||
|
||||
## Why a script rather than ctlptl
|
||||
|
||||
Deliberately a script rather than a tool. ctlptl collapses the `local` wiring
|
||||
into one line, but its Registry spec only accepts name/port/image/listenAddress —
|
||||
there is no way to set `REGISTRY_PROXY_REMOTEURL`, so it cannot express `mirror`
|
||||
at all. Keeping the seam here is what keeps the corporate registry swappable.
|
||||
|
||||
## CA trust (install_ca_into_nodes)
|
||||
|
||||
A corporate registry is almost always fronted by an internal CA, and trust has to
|
||||
reach three separate places. Nothing does this for you, and the symptom when it's
|
||||
missing is an opaque:
|
||||
|
||||
x509: certificate signed by unknown authority
|
||||
|
||||
1. the host docker daemon — `/etc/docker/certs.d/<host>/ca.crt` (needs root)
|
||||
2. every kind node's containerd — nodes do NOT inherit host trust
|
||||
3. anything doing HTTPS from inside the cluster, in its own trust store
|
||||
|
||||
`registry.sh` handles (2) because it's ours to handle. (1) is reported by
|
||||
`check.sh` since it needs root. (3) belongs to the workload.
|
||||
91
rig/docs/notes/selftest.md
Normal file
91
rig/docs/notes/selftest.md
Normal file
@@ -0,0 +1,91 @@
|
||||
# ctrl/selftest.sh
|
||||
|
||||
## Purpose
|
||||
|
||||
What rig has settled, written down as assertions.
|
||||
|
||||
These are documentation that runs. Each check is ONE decision that has already been made, with the reason above it: not coverage, and deliberately not an exhaustive sweep of use cases. rig's own index says a rule without its reason gets overridden the first time it is inconvenient; a rule nobody can restate is worse. So the test says what was decided, and failing it should read as "you are about to undo this" rather than "something broke".
|
||||
|
||||
Scope, on purpose:
|
||||
|
||||
- No cluster, no docker, no network. It must be cheap enough to actually run.
|
||||
- It asserts about RIG. `make check` asserts about the MACHINE and never fails; this exits 1, the way `make standalone check` does.
|
||||
- What actually deploys is not testable here. `tilt ci` stays a manual step.
|
||||
|
||||
## rig needs no profile
|
||||
|
||||
rig assumes no configuration. A profile is an overlay on built-in defaults, so a rig with no `env.d/` at all must resolve, report, and still generate a kit. Naming a profile that does not exist must still be an error, because a typo that silently fell back to the defaults would be worse than a failure.
|
||||
|
||||
## the ports.sh active contract
|
||||
|
||||
`ports.sh active` is read POSITIONALLY by two other files: the Makefile takes `$(word 2)` and `$(word 5)`, the Tiltfile takes `_facts[0]..[7]`. Insert a field in the middle and nothing errors: Tilt simply guards on the wrong context or binds the wrong port. The field count and order are the contract, so they are pinned here rather than left to whoever edits `ports.sh` next. The two paths are absolute or `-`, never empty: an empty field would shift the ones after it just the same.
|
||||
|
||||
## the caller's env beats the files
|
||||
|
||||
`lib/config.sh` states one precedence rule: `versions.env` < `env.d/<profile>` < `ctrl/.env` < the caller's env. It is enforced by `CONFIG_OVERRIDABLE`, a hand-maintained list, and a key missing from it loses to the file SILENTLY. `REGISTRY_PORT` and `MANIFESTS_DIR` were both missing on 2026-09-13 and were found by accident.
|
||||
|
||||
So the loop is generated FROM the list: add a key to `CONFIG_OVERRIDABLE` and the test starts asking about it without anyone remembering to come here. Three keys name something that must exist and are validated at load, so they get a real alternative rather than a sentinel.
|
||||
|
||||
## one derivation, not three
|
||||
|
||||
The Makefile used to compute the cluster name itself and sed `TILT_PORT` out of `ctrl/.env`: a second derivation of values `lib/config.sh` already owns, which could disagree with it after `ports.sh persist`. It now reads `ports.sh active`. Nothing structurally prevents the sed coming back, so the agreement is asserted against the real `make -n` output rather than against the source.
|
||||
|
||||
`--no-print-directory` and a grep, not `tail -1`: run from `make selftest` this is a RECURSIVE make, and the "Entering/Leaving directory" lines go to STDOUT. `tail -1` then reads "make[1]: Leaving directory ..." and both checks fail, but only when invoked through make, never when the script is run directly. A test that passes one way and fails the other is worse than no test.
|
||||
|
||||
## identity follows the folder, safely
|
||||
|
||||
The cluster name is NOT the bare directory name: kind needs a DNS label, so `default_cluster_name` lowercases it and replaces everything outside `[a-z0-9-]`. Re-deriving that anywhere else is how a copy ends up guarding the wrong context, which is exactly why the Tiltfile asks instead of computing.
|
||||
|
||||
## ports are stable across versions
|
||||
|
||||
Not a change-detector. The block is derived, never stored, so if the derivation shifts then every EXISTING environment's ports move underneath it: a running cluster keeps its old ports while rig starts reporting new ones, and `ports.sh show` stops describing reality. Anchored to three known names.
|
||||
|
||||
## rig stays standalone
|
||||
|
||||
rig sits inside a host project's tree but must be copyable straight out of it: no imports, no paths, no assumption the host is there. This grep is the whole test of that claim, and until it was added it lived only in prose and in whoever remembered to run it.
|
||||
|
||||
The pattern is assembled from fragments so the file does not match ITSELF. Writing it literally would fail forever; excluding the file instead would put a blind spot in the one check that guards the boundary. It includes the host project's word for a backing service, which rig's workload addons carried until they left, and skips `local/`, where overlays live and may say anything.
|
||||
|
||||
## scratch copies
|
||||
|
||||
Every check that changes something does it in a copy made by `copy_rig`: without `local/` (overlays, possibly someone else's, possibly large) and `def/`, and without this machine's `PROFILE`, `OVERLAY`, `CLUSTER` and `MANIFESTS_DIR` choices, so a check sets exactly what it tests.
|
||||
|
||||
## what runs is an overlay; rig only reads it
|
||||
|
||||
The overlay decisions (docs/notes/overlay.md), each against a throwaway overlay in a scratch copy: with nothing named, the same cluster, ports, addons, node count and kind config as before overlays existed; `rig.env` between the profile and `ctrl/.env`, the caller above all; identity from the overlay's folder; its paths relative to itself; a named overlay that does not exist, or a `rig.env` that tries to choose the profile or the overlay, is an error; an overlay's addon found before rig's own and run from rig's `ctrl/`; `persist` refusing; `make -n tilt OVERLAY=...` asking for the overlay's context (make before 4.4 would not pass it to `$(shell)`).
|
||||
|
||||
And the two that make an overlay safe to hold someone else's work: rig writes nothing into it (a checksum of the folder before and after `active`, `addons list`, a kind render, `standalone write` and `export`), and nothing from it — neither a value nor its path — reaches a committed kit.
|
||||
|
||||
## withdrawn stays withdrawn
|
||||
|
||||
One check per entry in STALE.md, each asserting that the withdrawn thing has not come back. The reasoning stays in STALE.md; the check is what makes it more than prose.
|
||||
|
||||
## the Tiltfile hardcodes nothing
|
||||
|
||||
Every other Tiltfile on this machine writes its slug in five or six times by hand, so a copied project deploys into the original's cluster until someone edits all of them. rig's asks `ports.sh`. A literal `kind-<name>` in it would mean that has been undone.
|
||||
|
||||
## standalone kits are generated and current
|
||||
|
||||
The kits under `standalone/<profile>/` are rig flattened into single files, one per profile. A kit left behind by a change to rig is exactly the drift they replaced (`rigmini.sh` once said 2 GB per node long after rig measured 800 MB), so a stale kit fails here rather than waiting to be noticed on another machine.
|
||||
|
||||
## kit Makefiles call only real verbs
|
||||
|
||||
Each kit's Makefile exists so nothing wrapping these scripts has to GUESS how to call them. A generated Makefile once did guess: `rigmini.sh on`, not a verb, and a bare `rigdeps.sh` for "check and report", which installs. So every target's default verb must be one its script's own dispatch accepts, read from that dispatch, not from a list that could drift from it.
|
||||
|
||||
## export carries choices, not credentials
|
||||
|
||||
An export is "take the setup I have here somewhere else", so it carries this machine's CHOICES (profile, ports, manifest dir) and never its credentials: `ctrl/.env` can hold registry and mirror logins next to those choices. The committed per-profile kits carry neither, since they must be the same on any machine. Proven with sentinel values in a scratch copy, because the real `ctrl/.env` may have those keys empty, and an empty value proves nothing.
|
||||
|
||||
## the dev loop parses — tilt and kubectl, no cluster
|
||||
|
||||
Parsing the Tiltfile for real is the only way to know it still evaluates. Tilt snapshots a kubectl context first, but it never contacts the cluster while evaluating: given a throwaway kubeconfig whose entries are kind-named (Tilt only runs `local()` freely for contexts it recognises as local) and a `kubectl` that swallows `apply`, `tilt alpha tiltfile-result` evaluates rig's Tiltfile with the starter overlay included, and reports the resources it would deploy.
|
||||
|
||||
It runs as `rig`, as a copy under another name — the case that used to stop at load (✖ S4) — and with the data overlay, whose namespace is used but not declared. Skipped, not failed, without tilt or kubectl.
|
||||
|
||||
## rig's addons apply verified files, never URLs
|
||||
|
||||
The offline example profile must install rig's addons with no network. Each addon therefore asks `deps.sh manifest <NAME>` for a pinned manifest, verified on disk, instead of applying a URL; the check fails if a URL comes back or a manifest is asked for without a pinned sum (versions.md). Their container images still need preloading, and nothing here pretends otherwise.
|
||||
|
||||
## the examples are overlays that work as shipped
|
||||
|
||||
`examples/` is what real overlays are copied from, so every addon there must parse, and every DAG must be valid Python. What they deploy is exercised by the parse checks above, not here.
|
||||
31
rig/docs/notes/standalone.md
Normal file
31
rig/docs/notes/standalone.md
Normal file
@@ -0,0 +1,31 @@
|
||||
# ctrl/standalone.sh
|
||||
|
||||
## Purpose
|
||||
|
||||
Generates the standalone kits: single-file versions of rig's own tools, one folder per profile, for machines the full rig is not going to.
|
||||
|
||||
A kit is a pure function of rig as it is right now. It gains nothing rig lacks and loses nothing rig has: improve rig, regenerate, and every kit follows. Nothing in `standalone/<profile>/` is ever edited by hand.
|
||||
|
||||
## The contract
|
||||
|
||||
What this file does NOT know, on purpose: which tools rig has, what they are called, how its libraries are split, where configuration lives or what it contains. Rig will change shape (scripts get split, renamed and grow new libraries), and a generator that encoded today's layout would quietly produce a wrong kit the first time it did. So it works from a contract a script opts into, and from nothing else:
|
||||
|
||||
1. A marker comment, alone on a line near the top, declares an entry point: `(hash) rig:standalone <kit-name> <default-verb>`. The default verb must only REPORT: it is run as a smoke test.
|
||||
2. Every `source` an entry point makes names a `.sh` file by a path that resolves relative to the entry point. Libraries may source further libraries however they like; bash follows those itself.
|
||||
3. Configuration enters through `load_config`, and the libraries provide `config_profiles`, `config_freeze <profile|--current>` (which prints a replacement `load_config` with that resolution frozen in) and, for an export, `config_current_profile` and `config_left_out`. How config is layered, stored, derived or frozen is rig's business; the generator only asks, and embeds the answer without interpreting it.
|
||||
|
||||
## Bash does the resolving
|
||||
|
||||
Bash does the resolving, not a parser in the generator. Libraries are sourced in a clean shell and read back with `declare -f` and `declare -p`, so any structure bash can load, this can flatten.
|
||||
|
||||
## Every kit is proven before it is written
|
||||
|
||||
Every kit is PROVEN to stand alone before it is written: no `source` left, no path into rig's tree in its code, `bash -n` clean, and its default verb run in an empty directory with nothing from rig present. A shape the generator has never seen either passes that, or generation stops and names the kit, the file, the line and what is wrong. It never writes a kit that only looks finished.
|
||||
|
||||
## Usage: write, check, export
|
||||
|
||||
- `standalone.sh write`: generate every kit into `standalone/<profile>/`.
|
||||
- `standalone.sh check`: generate into a scratch dir and fail if any kit differs.
|
||||
- `standalone.sh export DIR`: ONE kit for the configuration this machine runs (its profile plus the choices in its local config, WITHOUT its credentials), written outside the repo.
|
||||
|
||||
`write` and `check` are what gets committed: one kit per profile, identical on any machine. `export` answers the other question, "take the setup I have here somewhere else", so it reflects this machine, and for exactly that reason it never lands in the repository.
|
||||
54
rig/docs/notes/versions.md
Normal file
54
rig/docs/notes/versions.md
Normal file
@@ -0,0 +1,54 @@
|
||||
# ctrl/versions.env
|
||||
|
||||
## Pinned toolchain
|
||||
|
||||
The single manifest `ctrl/deps.sh` installs from. Every entry is a single binary; none of them needs an apt repo.
|
||||
|
||||
- kubectl — fully static
|
||||
- kind — libc only
|
||||
- tilt — libc + libstdc++ + libgcc (present in base Debian)
|
||||
- jq — upstream static build (Debian's is linked against libjq/libonig)
|
||||
|
||||
Checksums are the upstream-published SHA256 of the linux/amd64 artifact.
|
||||
|
||||
## Bumping a pin
|
||||
|
||||
Change the version, then take the checksum from the release's own published list — never hand-edit or hand-copy one from a download you did. For anything hosted on GitHub releases that is:
|
||||
|
||||
```
|
||||
curl -sSL https://github.com/<org>/<repo>/releases/download/<tag>/checksums.txt \
|
||||
| grep linux.x86_64
|
||||
```
|
||||
|
||||
(kubectl publishes its own instead: `<KUBECTL_URL>.sha256`.)
|
||||
|
||||
There was a `ctrl/versions-refresh.sh` named here that has never existed. If bumping stops being rare enough to do by hand, write it — but a comment pointing at a missing script is worse than no comment.
|
||||
|
||||
## ctlptl
|
||||
|
||||
Creates a kind cluster WITH a local registry wired in, which is what keeps images off docker.io (an unqualified name means `docker.io/library/<name>`). Same publisher and same archive shape as tilt: binary at the archive root, so `fetch_tgz` handles it with strip=0 and no special case.
|
||||
|
||||
## docker compose
|
||||
|
||||
The distro docker packages ship the daemon and the CLI but frequently not this, so `docker compose up` fails with "unknown command" on an otherwise working Docker. It is a CLI plugin, found by NAME in a plugin directory, so a copy in the bin dir alone only gives you the retired `docker-compose` v1 spelling; deps.sh links it into `~/.docker/cli-plugins`.
|
||||
|
||||
## Node images
|
||||
|
||||
Node images shipped with `KIND_VERSION`, pinned by digest so a kind upgrade can never silently move the k8s version. `K8S_VERSION` selects one (a profile or an overlay's rig.env may set it; the default is the newest pinned). Older entries are kept deliberately, for targets that run an older Kubernetes.
|
||||
|
||||
## Workload images are not pinned here
|
||||
|
||||
What an overlay runs is pinned by the overlay: `examples/data/rig.env` carries its postgres, redis and airflow images. This file holds what rig itself needs — the toolchain, the node images, the registry and rig's own addons — so it never says what any particular environment runs.
|
||||
|
||||
## The addons' manifests
|
||||
|
||||
metallb, cert-manager and metrics-server are installed from their upstream manifests. Those are pinned here by URL and SHA256 like the binaries, fetched through the same `DEPS_SOURCE` resolver (upstream, artifactory, baked) by `deps.sh manifest <NAME>`, verified, and applied from `vendor/manifests/` — never a URL applied directly. That is what lets the offline example profile install its addons with no network: the deps-full image carries them.
|
||||
|
||||
cert-manager and metrics-server publish their manifests as release assets, and GitHub reports each asset's SHA256 (`digest` in the release API); those are the pinned sums. metallb does not: its manifest is a file in the repository at the release tag, with no published sum. Its pin was taken from a download whose git blob id matched the one GitHub serves for `config/manifests/metallb-native.yaml` at that tag, and the blob id is kept beside it (`METALLB_MANIFEST_GIT_BLOB`) so the next bump is checked the same way:
|
||||
|
||||
```
|
||||
curl -sSL "https://api.github.com/repos/metallb/metallb/contents/config/manifests/metallb-native.yaml?ref=<tag>" | jq -r .sha
|
||||
(printf 'blob %d\0' "$(wc -c < metallb-native.yaml)"; cat metallb-native.yaml) | sha1sum
|
||||
```
|
||||
|
||||
Bumping an addon's version means bumping its manifest sum in the same edit; `deps.sh manifest` refuses a mismatch.
|
||||
98
rig/examples/data/README.md
Normal file
98
rig/examples/data/README.md
Normal file
@@ -0,0 +1,98 @@
|
||||
# `examples/data` — an overlay with a database, a scheduler and a worked pipeline
|
||||
|
||||
postgres, redis and airflow, each an upstream image run unmodified, installed as
|
||||
this overlay's own addons. rig ships the mechanism that finds and runs them; which
|
||||
services a workload needs is the workload's business, so they live here rather
|
||||
than in rig's `ctrl/addons/`. Copy what you need into your own overlay's `addons/`.
|
||||
|
||||
```bash
|
||||
OVERLAY=examples/data make cluster up # cluster `data`: postgres, then airflow
|
||||
OVERLAY=examples/data make tilt # the items-api simulator the DAG reads
|
||||
```
|
||||
|
||||
```
|
||||
rig.env ADDONS (metallb is rig's; the rest are here), namespace, identities, image pins
|
||||
addons/postgres.sh one replica on a PVC; the password generated once and kept
|
||||
addons/airflow.sh one `standalone` pod on its own `airflow` database; needs postgres
|
||||
addons/redis.sh only for switching airflow to CeleryExecutor (not in ADDONS)
|
||||
k8s/ the items-api simulator
|
||||
dags/items_to_postgres.py the worked example: API client → adapter → postgres
|
||||
Tiltfile names the simulator's resource
|
||||
```
|
||||
|
||||
Everything lands in the `data` namespace (`DATA_NAMESPACE`, and the kustomization
|
||||
names it too), so resetting an app's namespace leaves the databases alone. Costs
|
||||
roughly 2 GB with airflow, under 1 without. Airflow's first boot runs the whole
|
||||
metadata migration, so expect a few minutes before it is ready.
|
||||
|
||||
## The worked example: three links kept apart
|
||||
|
||||
- **Metadata DB (infra).** `addons/airflow.sh` creates an `airflow` database on the
|
||||
same postgres and points `SQL_ALCHEMY_CONN` there — airflow's own tables never land
|
||||
in the app's database.
|
||||
- **DAG delivery.** The addon turns this overlay's `dags/` into the `airflow-dags`
|
||||
ConfigMap, mounted at `/opt/airflow/dags`; re-run `make cluster up` after editing a
|
||||
DAG. The faster path later: a kind `extraMount` of `dags/` plus a Tilt `sync` — noted,
|
||||
not built.
|
||||
- **Data connection (operational logic).** `AIRFLOW_CONN_APP_DB`, composed each run
|
||||
from the postgres secret, gives DAGs the app's database as the `app_db` connection.
|
||||
One password reaches both URLs, with nowhere to drift.
|
||||
|
||||
The DAG itself calls the simulator with its own HTTP client (the wire, as the API
|
||||
returns it), renames the wire's fields into the app's names in `to_app_row` — **the
|
||||
adapter, which belongs to whoever owns the app's model and so lives in the overlay** —
|
||||
and upserts into `items`. Hourly, no backfill, one retry, idempotent on `item_id`.
|
||||
|
||||
```bash
|
||||
kubectl --context kind-data -n data exec deploy/airflow -- airflow dags unpause items_to_postgres
|
||||
kubectl --context kind-data -n data exec deploy/airflow -- airflow dags trigger items_to_postgres
|
||||
kubectl --context kind-data -n data exec deploy/postgres -- psql -U app -d app -c 'select * from items'
|
||||
```
|
||||
|
||||
## Addons in an overlay
|
||||
|
||||
Each one runs from rig's `ctrl/` (rig's `addons.sh` exports `RIG_CTRL`), so it
|
||||
starts with `cd "${RIG_CTRL:?...}"`, sources `./lib/config.sh` and calls
|
||||
`load_config` — and sees every key this overlay's `rig.env` sets. Run them through
|
||||
rig (`bash ctrl/addons.sh install`, or `make cluster up`), not directly.
|
||||
|
||||
## postgres — plain manifests, one replica
|
||||
|
||||
Plain manifests rather than a helm chart: a chart repo is a network dependency,
|
||||
and an offline machine needs a path with none. The image is pinned in `rig.env`
|
||||
and can be preloaded into a local registry like every other image.
|
||||
|
||||
One replica on a PVC. This models a dependency for local work, not a
|
||||
highly-available database, and pretending otherwise on a kind node would be a
|
||||
more elaborate lie rather than a more useful one.
|
||||
|
||||
The password is not in `rig.env`: `addons/postgres.sh` generates one on first
|
||||
install and keeps it across re-runs, so re-running the addon never rotates the
|
||||
credential out from under whatever is already connected.
|
||||
|
||||
## redis
|
||||
|
||||
Cache, and the broker anything queue-shaped runs on. No persistence: a broker
|
||||
that loses its queue on restart is the honest local model, and a PVC here buys
|
||||
nothing but a volume to clean up.
|
||||
|
||||
## airflow
|
||||
|
||||
Airflow needs a metadata database before it will start at all, so the script
|
||||
refuses rather than rolls a pod that will CrashLoopBackOff while the real problem
|
||||
(postgres missing from `ADDONS`) stays invisible in the logs.
|
||||
|
||||
One pod on `standalone`: migration, admin user, scheduler and webserver in a
|
||||
single container, on LocalExecutor, which needs no broker. The official chart's
|
||||
five deployments model an installation; switching this on means wanting pipelines.
|
||||
redis is here for the day it moves to CeleryExecutor, and not before.
|
||||
|
||||
## Reaching them
|
||||
|
||||
Reach the databases with port-forward rather than binding more host ports:
|
||||
|
||||
```bash
|
||||
kubectl --context kind-data -n data port-forward svc/postgres 5432:5432
|
||||
kubectl --context kind-data -n data port-forward svc/airflow 8080:8080
|
||||
kubectl --context kind-data -n data get secret postgres -o jsonpath='{.data.POSTGRES_PASSWORD}' | base64 -d
|
||||
```
|
||||
6
rig/examples/data/Tiltfile
Normal file
6
rig/examples/data/Tiltfile
Normal file
@@ -0,0 +1,6 @@
|
||||
# The data overlay's half of the dev loop: the items-api simulator the example DAG reads.
|
||||
# rig's ctrl/Tiltfile has already applied k8s/overlays/dev; paths here are relative to
|
||||
# this folder. postgres and airflow are addons (make cluster up), not Tilt resources.
|
||||
# Notes: README.md
|
||||
|
||||
k8s_resource('items-api', labels=['simulator'])
|
||||
@@ -1,15 +1,11 @@
|
||||
#!/usr/bin/env bash
|
||||
# Apache Airflow — the cluster half of the airflow cabinet.
|
||||
#
|
||||
# Airflow needs a metadata database before it will start at all, so this refuses
|
||||
# rather than rolls a pod that will CrashLoopBackOff while the real problem
|
||||
# (postgres missing from ADDONS) stays invisible in the logs.
|
||||
#
|
||||
# One pod on `standalone`, matching the compose cabinet: migration, admin user,
|
||||
# scheduler and webserver in a single container. The official chart's five
|
||||
# deployments model an installation; switching this on means wanting pipelines.
|
||||
# Apache Airflow for this overlay: one `standalone` pod (LocalExecutor — no broker).
|
||||
# Its own `airflow` database on the postgres addon; the app's data reaches DAGs as the
|
||||
# `app_db` connection; DAGs from this overlay's dags/, as a ConfigMap.
|
||||
# Requires the postgres addon; refuses to install without it.
|
||||
# Notes: ../README.md
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")/.."
|
||||
cd "${RIG_CTRL:?run it through rig: bash ctrl/addons.sh install}"
|
||||
|
||||
source ./lib/config.sh
|
||||
load_config
|
||||
@@ -19,16 +15,25 @@ NS="${DATA_NAMESPACE:-data}"
|
||||
|
||||
if ! $K get deployment -n "$NS" postgres >/dev/null 2>&1; then
|
||||
echo " ! airflow needs the postgres addon, and it is not installed" >&2
|
||||
echo " add it before airflow in the profile's ADDONS:" >&2
|
||||
echo " add it before airflow in the overlay's ADDONS:" >&2
|
||||
echo " ADDONS=\"... postgres airflow\"" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Reuse the credential postgres generated rather than storing a second copy.
|
||||
# ── metadata DB: airflow's own tables, kept out of the app's database ──────
|
||||
# Same postgres instance, separate database, created once (idempotent).
|
||||
db_user=$($K get secret -n "$NS" postgres -o jsonpath='{.data.POSTGRES_USER}' | base64 -d)
|
||||
db_pass=$($K get secret -n "$NS" postgres -o jsonpath='{.data.POSTGRES_PASSWORD}' | base64 -d)
|
||||
db_name=$($K get secret -n "$NS" postgres -o jsonpath='{.data.POSTGRES_DB}' | base64 -d)
|
||||
psql() { $K exec -n "$NS" deploy/postgres -- psql -U "$db_user" -d "$db_name" -tAc "$1"; }
|
||||
if [ "$(psql "SELECT 1 FROM pg_database WHERE datname = 'airflow'")" = 1 ]; then
|
||||
echo " database 'airflow' exists"
|
||||
else
|
||||
psql "CREATE DATABASE airflow" >/dev/null
|
||||
echo " created database 'airflow' beside '${db_name}'"
|
||||
fi
|
||||
|
||||
# ── what is generated once and kept: re-running never rotates these ────────
|
||||
if $K get secret -n "$NS" airflow >/dev/null 2>&1; then
|
||||
echo " secret exists, keeping the current admin password and fernet key"
|
||||
else
|
||||
@@ -40,11 +45,30 @@ else
|
||||
--from-literal=ADMIN_USER="${AIRFLOW_ADMIN_USER:-admin}" \
|
||||
--from-literal=ADMIN_PASSWORD="$admin_password" \
|
||||
--from-literal=FERNET_KEY="$fernet_key" \
|
||||
--from-literal=SQL_ALCHEMY_CONN="postgresql+psycopg2://${db_user}:${db_pass}@postgres:5432/${db_name}" \
|
||||
>/dev/null
|
||||
echo " generated an admin password (read it back with the command below)"
|
||||
fi
|
||||
|
||||
# ── connections: composed from what the postgres secret owns, every run ─────
|
||||
# Nothing to drift: one password reaches the metadata DB and the data connection.
|
||||
$K create secret generic airflow-connections -n "$NS" \
|
||||
--from-literal=SQL_ALCHEMY_CONN="postgresql+psycopg2://${db_user}:${db_pass}@postgres:5432/airflow" \
|
||||
--from-literal=AIRFLOW_CONN_APP_DB="postgres://${db_user}:${db_pass}@postgres:5432/${db_name}" \
|
||||
--dry-run=client -o yaml | $K apply -f - >/dev/null
|
||||
|
||||
# ── DAG delivery: this overlay's dags/ as a ConfigMap ───────────────────────
|
||||
# Edits land by re-running this addon (make cluster up). The later path — a kind
|
||||
# extraMount of dags/ plus a Tilt sync — is noted in the README, not built.
|
||||
# A ConfigMap volume is kubelet's ..data/..<timestamp> symlinks, and Airflow's DAG walker
|
||||
# follows symlinks: without the .airflowignore it stops at "Detected recursive loop".
|
||||
dags="$(_from_ctrl "$OVERLAY_DIR")/dags"
|
||||
if [ -d "$dags" ]; then
|
||||
$K create configmap airflow-dags -n "$NS" --from-file="$dags" \
|
||||
--from-literal=.airflowignore='^\.\.' \
|
||||
--dry-run=client -o yaml | $K apply -f - >/dev/null
|
||||
echo " dags: $(ls "$dags" | grep -c '\.py$') file(s) from $(basename "$(_abs_from_ctrl "$OVERLAY_DIR")")/dags"
|
||||
fi
|
||||
|
||||
echo " applying manifests"
|
||||
$K apply -n "$NS" -f - >/dev/null <<YAML
|
||||
apiVersion: v1
|
||||
@@ -85,7 +109,10 @@ spec:
|
||||
value: "false"
|
||||
- name: AIRFLOW__DATABASE__SQL_ALCHEMY_CONN
|
||||
valueFrom:
|
||||
secretKeyRef: {name: airflow, key: SQL_ALCHEMY_CONN}
|
||||
secretKeyRef: {name: airflow-connections, key: SQL_ALCHEMY_CONN}
|
||||
- name: AIRFLOW_CONN_APP_DB
|
||||
valueFrom:
|
||||
secretKeyRef: {name: airflow-connections, key: AIRFLOW_CONN_APP_DB}
|
||||
- name: AIRFLOW__CORE__FERNET_KEY
|
||||
valueFrom:
|
||||
secretKeyRef: {name: airflow, key: FERNET_KEY}
|
||||
@@ -97,6 +124,9 @@ spec:
|
||||
secretKeyRef: {name: airflow, key: ADMIN_PASSWORD}
|
||||
ports:
|
||||
- containerPort: 8080
|
||||
volumeMounts:
|
||||
- name: dags
|
||||
mountPath: /opt/airflow/dags
|
||||
readinessProbe:
|
||||
httpGet:
|
||||
path: /health
|
||||
@@ -105,6 +135,11 @@ spec:
|
||||
initialDelaySeconds: 60
|
||||
periodSeconds: 15
|
||||
failureThreshold: 20
|
||||
volumes:
|
||||
- name: dags
|
||||
configMap:
|
||||
name: airflow-dags
|
||||
optional: true
|
||||
YAML
|
||||
|
||||
echo " waiting for airflow (the first boot migrates the database, so this is slow)..."
|
||||
@@ -1,22 +1,8 @@
|
||||
#!/usr/bin/env bash
|
||||
# PostgreSQL — the cluster half of the postgres cabinet.
|
||||
#
|
||||
# A cabinet is a public service dropped into the environment as-is — the
|
||||
# upstream image, unmodified, reachable at a known address. This is the cluster
|
||||
# half of it; the compose half is a `service.yml` beside a `cabinet.json`. The
|
||||
# declaration is made once and both paths read it, so nothing is remembered
|
||||
# twice.
|
||||
#
|
||||
# Plain manifests rather than a helm chart, matching the other addons: a chart
|
||||
# repo is a network dependency, and the offline profile exists precisely so
|
||||
# there is a path with none. The image is pinned in ctrl/versions.env and can be
|
||||
# preloaded into a local registry like every other image here.
|
||||
#
|
||||
# One replica on a PVC. This models a dependency for local work, not a
|
||||
# highly-available database, and pretending otherwise on a kind node would be a
|
||||
# more elaborate lie rather than a more useful one.
|
||||
# PostgreSQL for this overlay: plain manifests, one replica on a PVC, password generated once and kept.
|
||||
# Notes: ../README.md
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")/.."
|
||||
cd "${RIG_CTRL:?run it through rig: bash ctrl/addons.sh install}"
|
||||
|
||||
source ./lib/config.sh
|
||||
load_config
|
||||
@@ -1,11 +1,8 @@
|
||||
#!/usr/bin/env bash
|
||||
# Redis — the cluster half of the redis cabinet.
|
||||
#
|
||||
# Cache, and the broker anything queue-shaped runs on. No persistence: a broker
|
||||
# that loses its queue on restart is the honest local model, and a PVC here buys
|
||||
# nothing but a volume to clean up.
|
||||
# Redis for this overlay: cache and broker (Celery), no persistence.
|
||||
# Notes: ../README.md
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")/.."
|
||||
cd "${RIG_CTRL:?run it through rig: bash ctrl/addons.sh install}"
|
||||
|
||||
source ./lib/config.sh
|
||||
load_config
|
||||
81
rig/examples/data/dags/items_to_postgres.py
Normal file
81
rig/examples/data/dags/items_to_postgres.py
Normal file
@@ -0,0 +1,81 @@
|
||||
"""Pull items from the items API, rename them into the app's names, upsert into postgres.
|
||||
|
||||
Three links, kept apart on purpose:
|
||||
|
||||
- the API client (`fetch_items`) talks to the wire as it is — here the overlay's
|
||||
own simulator, `items-api`, whose field names are the API's;
|
||||
- the adapter (`to_app_row`) is the one place the wire's names become the app's:
|
||||
`id` -> `item_id`, `name` -> `item_name`, `price.amount_cents` -> `price_cents`.
|
||||
It belongs to whoever owns the app's model, so it lives in the overlay, not in rig;
|
||||
- the load writes through the `app_db` connection (AIRFLOW_CONN_APP_DB, built by
|
||||
addons/airflow.sh from the postgres secret) and is idempotent: an upsert keyed
|
||||
on `item_id`, so a retry or a rerun never duplicates a row.
|
||||
|
||||
Operational logic is explicit and minimal: hourly, no backfill, one retry.
|
||||
"""
|
||||
import json
|
||||
import urllib.request
|
||||
from datetime import datetime, timedelta
|
||||
|
||||
from airflow import DAG
|
||||
from airflow.operators.python import PythonOperator
|
||||
|
||||
ITEMS_URL = "http://items-api/v1/items"
|
||||
|
||||
CREATE = """
|
||||
CREATE TABLE IF NOT EXISTS items (
|
||||
item_id text PRIMARY KEY,
|
||||
item_name text NOT NULL,
|
||||
price_cents integer NOT NULL,
|
||||
currency text NOT NULL,
|
||||
loaded_at timestamptz NOT NULL DEFAULT now()
|
||||
)
|
||||
"""
|
||||
|
||||
UPSERT = """
|
||||
INSERT INTO items (item_id, item_name, price_cents, currency)
|
||||
VALUES (%(item_id)s, %(item_name)s, %(price_cents)s, %(currency)s)
|
||||
ON CONFLICT (item_id) DO UPDATE
|
||||
SET item_name = EXCLUDED.item_name,
|
||||
price_cents = EXCLUDED.price_cents,
|
||||
currency = EXCLUDED.currency,
|
||||
loaded_at = now()
|
||||
"""
|
||||
|
||||
|
||||
def fetch_items():
|
||||
"""The API client: the wire, as the API returns it."""
|
||||
with urllib.request.urlopen(ITEMS_URL, timeout=10) as response:
|
||||
return json.load(response)["items"]
|
||||
|
||||
|
||||
def to_app_row(item):
|
||||
"""The adapter: the API's names in, the app's names out."""
|
||||
return {
|
||||
"item_id": item["id"],
|
||||
"item_name": item["name"],
|
||||
"price_cents": item["price"]["amount_cents"],
|
||||
"currency": item["price"]["currency"],
|
||||
}
|
||||
|
||||
|
||||
def load_items():
|
||||
from airflow.providers.postgres.hooks.postgres import PostgresHook
|
||||
|
||||
rows = [to_app_row(item) for item in fetch_items()]
|
||||
hook = PostgresHook(postgres_conn_id="app_db")
|
||||
hook.run(CREATE)
|
||||
for row in rows:
|
||||
hook.run(UPSERT, parameters=row)
|
||||
print(f"upserted {len(rows)} items")
|
||||
|
||||
|
||||
with DAG(
|
||||
dag_id="items_to_postgres",
|
||||
schedule="@hourly",
|
||||
start_date=datetime(2026, 1, 1),
|
||||
catchup=False,
|
||||
default_args={"retries": 1, "retry_delay": timedelta(minutes=1)},
|
||||
tags=["example"],
|
||||
) as dag:
|
||||
PythonOperator(task_id="load_items", python_callable=load_items)
|
||||
95
rig/examples/data/k8s/base/items-api.yaml
Normal file
95
rig/examples/data/k8s/base/items-api.yaml
Normal file
@@ -0,0 +1,95 @@
|
||||
# The simulator: a stub of the API the DAG reads, faithful to the wire (its field
|
||||
# names are the API's, not the app's). Same shape as the starter's example-mock.
|
||||
apiVersion: v1
|
||||
kind: ConfigMap
|
||||
metadata:
|
||||
name: items-api-stub
|
||||
data:
|
||||
routes.json: |
|
||||
{
|
||||
"/health": {"status": 200, "body": {"status": "ok"}},
|
||||
"/v1/items": {"status": 200, "body": {"items": [
|
||||
{"id": "a-100", "name": "anvil", "price": {"amount_cents": 1999, "currency": "USD"}},
|
||||
{"id": "b-200", "name": "bucket", "price": {"amount_cents": 450, "currency": "USD"}},
|
||||
{"id": "c-300", "name": "crate", "price": {"amount_cents": 1200, "currency": "USD"}}
|
||||
]}}
|
||||
}
|
||||
serve.py: |
|
||||
import json, os
|
||||
from http.server import BaseHTTPRequestHandler, HTTPServer
|
||||
|
||||
ROUTES = json.load(open("/etc/stub/routes.json"))
|
||||
NAME = os.environ.get("STUB_NAME", "stub")
|
||||
|
||||
class H(BaseHTTPRequestHandler):
|
||||
def do_GET(self):
|
||||
r = ROUTES.get(self.path)
|
||||
if r is None:
|
||||
self.send_response(404)
|
||||
self.end_headers()
|
||||
self.wfile.write(json.dumps(
|
||||
{"error": "no canned route", "stub": NAME, "path": self.path}
|
||||
).encode())
|
||||
return
|
||||
body = json.dumps(r["body"]).encode()
|
||||
self.send_response(r["status"])
|
||||
self.send_header("Content-Type", "application/json")
|
||||
self.send_header("X-Mocked-By", NAME)
|
||||
self.end_headers()
|
||||
self.wfile.write(body)
|
||||
|
||||
def log_message(self, fmt, *args):
|
||||
print("%s %s" % (NAME, fmt % args), flush=True)
|
||||
|
||||
HTTPServer(("0.0.0.0", 8080), H).serve_forever()
|
||||
---
|
||||
apiVersion: apps/v1
|
||||
kind: Deployment
|
||||
metadata:
|
||||
name: items-api
|
||||
labels:
|
||||
app: items-api
|
||||
rig.component/impl: mock
|
||||
spec:
|
||||
replicas: 1
|
||||
selector:
|
||||
matchLabels:
|
||||
app: items-api
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
app: items-api
|
||||
spec:
|
||||
containers:
|
||||
- name: stub
|
||||
image: python:3.12-slim
|
||||
command: ["python3", "/etc/stub/serve.py"]
|
||||
env:
|
||||
- name: STUB_NAME
|
||||
value: items-api
|
||||
ports:
|
||||
- containerPort: 8080
|
||||
volumeMounts:
|
||||
- name: stub
|
||||
mountPath: /etc/stub
|
||||
readinessProbe:
|
||||
httpGet: { path: /health, port: 8080 }
|
||||
initialDelaySeconds: 2
|
||||
resources:
|
||||
requests: { memory: 32Mi, cpu: 10m }
|
||||
limits: { memory: 64Mi }
|
||||
volumes:
|
||||
- name: stub
|
||||
configMap:
|
||||
name: items-api-stub
|
||||
---
|
||||
apiVersion: v1
|
||||
kind: Service
|
||||
metadata:
|
||||
name: items-api
|
||||
spec:
|
||||
selector:
|
||||
app: items-api
|
||||
ports:
|
||||
- port: 80
|
||||
targetPort: 8080
|
||||
10
rig/examples/data/k8s/base/kustomization.yaml
Normal file
10
rig/examples/data/k8s/base/kustomization.yaml
Normal file
@@ -0,0 +1,10 @@
|
||||
apiVersion: kustomize.config.k8s.io/v1beta1
|
||||
kind: Kustomization
|
||||
|
||||
# Beside the addons, in DATA_NAMESPACE — but no Namespace object: the addons created
|
||||
# it, and a Namespace Tilt owned would be deleted by `tilt down`, taking postgres and
|
||||
# airflow with it. rig's Tiltfile creates namespaces that are used and not declared.
|
||||
namespace: data
|
||||
|
||||
resources:
|
||||
- items-api.yaml
|
||||
5
rig/examples/data/k8s/overlays/dev/kustomization.yaml
Normal file
5
rig/examples/data/k8s/overlays/dev/kustomization.yaml
Normal file
@@ -0,0 +1,5 @@
|
||||
apiVersion: kustomize.config.k8s.io/v1beta1
|
||||
kind: Kustomization
|
||||
|
||||
resources:
|
||||
- ../../base
|
||||
23
rig/examples/data/rig.env
Normal file
23
rig/examples/data/rig.env
Normal file
@@ -0,0 +1,23 @@
|
||||
# This overlay's settings: layered over rig's defaults, under ctrl/.env and the caller.
|
||||
# data — postgres, redis and airflow (upstream images, run unmodified) in their own namespace.
|
||||
# Use it: OVERLAY=examples/data make cluster up Notes: README.md
|
||||
|
||||
# Order matters: addons install in the order listed, and airflow refuses to start
|
||||
# without postgres, so postgres comes first. metallb is rig's own. Airflow runs
|
||||
# LocalExecutor and needs no broker: add redis (before airflow) only to switch to Celery.
|
||||
ADDONS="metallb postgres airflow"
|
||||
|
||||
# Namespace for the dependency containers (k8s/base/kustomization.yaml names it too).
|
||||
DATA_NAMESPACE=data
|
||||
|
||||
# Postgres identity. The password is generated once by addons/postgres.sh and kept.
|
||||
POSTGRES_DB=app
|
||||
POSTGRES_USER=app
|
||||
POSTGRES_STORAGE=2Gi
|
||||
|
||||
AIRFLOW_ADMIN_USER=admin
|
||||
|
||||
# Upstream images, pinned by tag; bump freely, and preload them for an offline machine.
|
||||
POSTGRES_IMAGE=postgres:16-alpine
|
||||
REDIS_IMAGE=redis:7-alpine
|
||||
AIRFLOW_IMAGE=apache/airflow:2.10.4
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user