diff --git a/Makefile b/Makefile index 739311c..7827c36 100644 --- a/Makefile +++ b/Makefile @@ -38,7 +38,7 @@ workspace-build: rm -f $(WORKSPACE_OVERRIDE); \ fi -.PHONY: help generate init up down restart clean nuke fresh logs pull build cli status guppy regen debug-upload ensure-state check-docker workspace-build shell-guppy shell-piri shell-upload shell-hilt +.PHONY: help generate init up down restart clean nuke fresh logs pull build cli status guppy regen debug-upload ensure-state check-docker workspace-build shell-guppy shell-piri shell-upload shell-hilt staging-keygen staging-bootstrap staging-provision-core staging-provision-piri staging-vault-init staging-deploy-core staging-allowlist-piri staging-deploy-piri staging-register-piri staging-register-ingot staging-fund-payer staging-reset # Default target - show help help: @@ -82,6 +82,20 @@ help: @echo "Debugging:" @echo " make debug-upload Run upload (sprue) under Delve on localhost:2345" @echo "" + @echo "Staging deployment (see docs/STAGING_DEPLOY.md):" + @echo " make staging-keygen Ensure keys+wallets+proofs exist in 1Password (idempotent)" + @echo " make staging-bootstrap One-time: prepare the box (repo, dirs, Caddy)" + @echo " make staging-provision-core Render core secrets from 1Password and ship to the box (dev machine only)" + @echo " make staging-provision-piri Render piri secrets from 1Password and ship to the box (dev machine only)" + @echo " make staging-vault-init Init + unseal hilt-vault; store keys in 1Password (run between provision-core and deploy-core)" + @echo " make staging-deploy-core Deploy the core bundle (sprue + signing-service + delegator + hilt + plc)" + @echo " make staging-allowlist-piri Allow-list the piri DID with the delegator (run before deploy-piri)" + @echo " make staging-deploy-piri Deploy the piri bundle (piri-0 + ingot)" + @echo " make staging-register-piri Register piri as a storage provider with sprue (run after both deploys)" + @echo " make staging-register-ingot Register ingot as hilt's regional provider (run after both deploys)" + @echo " make staging-fund-payer Deposit USDFC into FilecoinPay so piri can create a proof set" + @echo " make staging-reset Wipe ALL staging data and redeploy both bundles from scratch (destructive; keeps keys+wallets)" + @echo "" @echo "Options:" @echo " YES=1 Skip confirmation prompts (e.g., make nuke YES=1)" @echo " SMELT_WORKSPACE=1 Run containers against binaries built from your local" @@ -260,6 +274,95 @@ regen: @echo "Keys and proofs regenerated." @echo "Run 'make clean && make up' to restart services with new keys." +# --- Staging deployment --------------------------------------------------- +# Thin wrappers over the staging tooling; full runbook in docs/STAGING_DEPLOY.md. + +# Ensure staging keys, EVM wallets, and UCAN proofs exist (idempotent: existing +# 1Password fields are reused, only missing ones are generated); write proofs to +# environments/staging/proofs/ (commit them); write wallet addresses (incl. +# PAYER_ADDRESS) into environments/staging/wallets.env (commit it). +staging-keygen: + @go run ./cmd/smelt staging keygen + +# One-time: prepare the box — clone/update the repo, create secrets + data dirs, +# wire the Caddy snippet, verify. Pass REPO_URL on first run. +staging-bootstrap: + @./scripts/staging-bootstrap.sh + +# Render configs/keys from 1Password and stream them to the box. Developer +# machine only (needs your op session + SSH); never run from CI. Per-bundle, since +# we typically deploy one bundle at a time. +staging-provision-core: + @./scripts/staging-provision.sh core + +staging-provision-piri: + @./scripts/staging-provision.sh piri + +# Initialize + unseal the persistent (Raft) hilt-vault and store its unseal key + +# root token in 1Password. Run after provision-core and before deploy-core: the +# unseal key/root token are minted at runtime by `vault operator init` (not by +# keygen) and rendered into vault-secrets.env, which deploy-core consumes. +# Developer machine only (needs your op session + jq). Idempotent. +staging-vault-init: + @./scripts/staging-vault-init.sh + +# Deploy a bundle: pull pinned images, recreate, verify health. +staging-deploy-core: + @./scripts/staging-deploy.sh core + +# Allow-list the piri DID with the core delegator. Run after deploy-core and +# before deploy-piri, else `piri init` step [4/7] gets a 403 from the delegator. +staging-allowlist-piri: + @./scripts/staging-allowlist-piri.sh + +staging-deploy-piri: + @./scripts/staging-deploy.sh piri + +# Register the piri node as a storage provider with sprue. Run after BOTH +# bundles are healthy, else uploads fail with "no storage providers available" +# (locally sprue's post_start hook does this; across bundles it can't). +staging-register-piri: + @./scripts/staging-register-piri.sh + +# Register ingot as hilt's regional provider. Run after BOTH bundles are healthy +# and BEFORE creating any tenants, else the Tenant API rejects the region +# (locally hilt's post_start hook does this; across bundles it can't). +staging-register-ingot: + @./scripts/staging-register-ingot.sh + +# Deposit USDFC into FilecoinPay for the payer + grant the warm-storage service +# operator approval, so `piri init`'s proof-set creation clears +# InsufficientLockupFunds. Developer machine only (needs your op session + cast). +# Amounts are baked in but env-overridable; see scripts/staging-fund-payer.sh. +staging-fund-payer: + @./scripts/staging-fund-payer.sh + +# Full destructive reset of BOTH staging bundles in one shot: wipe all data +# (Hilt tenants + access keys, piri objects, ingot buckets + blob_locations, +# Postgres, DynamoDB, MinIO, the hilt-vault Raft store) and redeploy from +# scratch. Runs the whole provision -> vault-init -> deploy -> register sequence +# in the correct order, aborting on the first failure. +# +# What SURVIVES: all Ed25519 service identities and EVM wallet addresses (only +# re-shipped from 1Password, never regenerated) and all on-chain state (wallet +# balances, the payer's FilecoinPay deposit). What ROTATES: the hilt-vault unseal +# key + root token, since the Vault store is wiped and `vault operator init` +# mints them at runtime (staging-vault-init overwrites the two 1Password fields). +# +# Set FORGE_REF to deploy a branch other than main: +# FORGE_REF=staging-deployment make staging-reset +# Developer machine only (needs your op session + SSH); never run from CI. +staging-reset: + @$(MAKE) --no-print-directory staging-provision-core + @$(MAKE) --no-print-directory staging-provision-piri + @$(MAKE) --no-print-directory staging-vault-init + @$(MAKE) --no-print-directory staging-deploy-core + @$(MAKE) --no-print-directory staging-allowlist-piri + @$(MAKE) --no-print-directory staging-deploy-piri + @$(MAKE) --no-print-directory staging-register-piri + @$(MAKE) --no-print-directory staging-register-ingot + @echo "staging-reset complete." + # Pull latest pre-built images (ignores failures for local-only images) pull: generated/compose/piri.yml ensure-state $(COMPOSE) pull --ignore-pull-failures diff --git a/cmd/smelt/cmd/staging.go b/cmd/smelt/cmd/staging.go new file mode 100644 index 0000000..2f855f1 --- /dev/null +++ b/cmd/smelt/cmd/staging.go @@ -0,0 +1,110 @@ +package cmd + +import ( + "fmt" + + "github.com/fil-forge/smelt/pkg/staging" + "github.com/spf13/cobra" +) + +var stagingCmd = &cobra.Command{ + Use: "staging", + Short: "Manage the Forge staging deployment", +} + +var stagingKeygenCmd = &cobra.Command{ + Use: "keygen", + Short: "Ensure staging keys, wallets, and proofs exist (idempotent)", + Long: `Ensures the staging stack's long-lived secrets exist in 1Password and writes +the (non-secret) UCAN delegation proofs into environments/staging/proofs/ to be +committed. + +The ceremony is idempotent: every field the 1Password item already holds is +reused byte-for-byte — funded wallets, registered DIDs, and shipped keys survive +a re-run — and only missing fields are generated and added. Proofs are re-issued +only when missing or when a key they depend on was freshly generated. + +It covers: + - Ed25519 service identity keys (PEM), incl. hilt and ingot + - real random secp256k1 EVM wallets for the payer, delegator transactor, and + piri owner — their addresses are printed so you can fund them via a Calibnet + faucet (private keys are stored in 1Password, never printed) + - random connection secrets (Postgres admin + per-service passwords, MinIO + keys, hilt partner key, ingot root S3 credentials) + - the indexing/egress/piri/hilt/ingot UCAN delegation proofs + +The hilt Vault unseal key + root token are NOT minted here — they are produced +by "vault operator init" at runtime and stored in 1Password by +"make staging-vault-init". + +To rotate a specific secret, delete its field from the 1Password item (and any +proof files signed with it) and re-run.`, + RunE: runStagingKeygen, +} + +func init() { + rootCmd.AddCommand(stagingCmd) + stagingCmd.AddCommand(stagingKeygenCmd) + stagingKeygenCmd.Flags().StringP("project-dir", "d", ".", "project root directory") + stagingKeygenCmd.Flags().String("op-vault", "Fil One", "1Password vault") + stagingKeygenCmd.Flags().String("op-item", "FilOne Forge Staging", "1Password item title") + stagingKeygenCmd.Flags().Bool("store", true, "store generated secrets into 1Password via the op CLI") + stagingKeygenCmd.Flags().Bool("proofs", true, "generate UCAN delegation proofs (requires ucantool)") + stagingKeygenCmd.Flags().String("ucantool", "ucantool", "ucantool binary name or path") +} + +func runStagingKeygen(cmd *cobra.Command, args []string) error { + projectDir, _ := cmd.Flags().GetString("project-dir") + opVault, _ := cmd.Flags().GetString("op-vault") + opItem, _ := cmd.Flags().GetString("op-item") + store, _ := cmd.Flags().GetBool("store") + proofs, _ := cmd.Flags().GetBool("proofs") + ucantool, _ := cmd.Flags().GetString("ucantool") + + result, err := staging.Keygen(staging.Options{ + ProjectDir: projectDir, + OPVault: opVault, + OPItem: opItem, + Store: store, + Proofs: proofs, + Ucantool: ucantool, + }) + if err != nil { + return err + } + + fmt.Println("Staging keygen complete.") + if len(result.GeneratedFields) > 0 { + fmt.Printf("\nNewly generated field(s):\n") + for _, f := range result.GeneratedFields { + fmt.Printf(" %s\n", f) + } + } + if len(result.ReusedFields) > 0 { + fmt.Printf("\nReused %d existing field(s) from 1Password (not rotated).\n", len(result.ReusedFields)) + } + if len(result.GeneratedFields) == 0 && len(result.ReusedFields) > 0 { + fmt.Println("All secrets already existed — nothing was rotated.") + } + if len(result.ProofsWritten) > 0 { + fmt.Printf("\nProofs written (commit these):\n") + for _, p := range result.ProofsWritten { + fmt.Printf(" %s\n", p) + } + } + if len(result.OPFields) > 0 { + fmt.Printf("\nStored %d secret field(s) in 1Password item %q (vault %q):\n", + len(result.OPFields), opItem, opVault) + for _, f := range result.OPFields { + fmt.Printf(" %s\n", f) + } + } + if result.WalletsEnvPath != "" { + fmt.Printf("\nWrote wallet addresses to %s (commit it).\n", result.WalletsEnvPath) + } + fmt.Printf("\nFund these wallets on the Calibnet faucet (https://faucet.calibnet.chainsafe-fil.io):\n") + for role, addr := range result.FundAddresses { + fmt.Printf(" %-24s %s\n", role+":", addr) + } + return nil +} diff --git a/docs/STAGING_DEPLOY.md b/docs/STAGING_DEPLOY.md new file mode 100644 index 0000000..884c15b --- /dev/null +++ b/docs/STAGING_DEPLOY.md @@ -0,0 +1,757 @@ +# Staging Deployment Runbook + +How to manually deploy the Forge stack to the **staging** box. + +> **Scope (first step):** manual deploy, no CI. Two independently-deployed Compose +> bundles on one VM, secrets injected from 1Password at provision time. + +## Rationale + +The goal is to get the staging deployment up as quickly as possible. We are optimising for learning +and discovering unknown unknowns. See [Open Questions in the Walking Skeleton Notion page](https://app.notion.com/p/filecoin/FilOne-SP-Side-Appliance-Walking-Skeleton-3847631f2825806f876ac0fd478e5a68?source=copy_link#3847631f2825806d90bcc385affd9659). + +The two-bundle split is deliberate to force decoupling between Forge core services (operated by +FilOne in production) and storage nodes (operated by storage providers in production). + +## Architecture + +Forge services are each released independently to GHCR. Staging composes them into +Compose manifests held in this repo — the unit of deployment. The target design (once +deployments are automated) is for these manifests to be **version-pinned** to exact image +digests. In this initial version, where we deploy manually, they instead track the rolling +`:main` tag of each service. The design rests on four ideas: + +1. **One version-pinned artifact per bundle (future goal).** The aim is for each bundle's + `versions.env` to pin exact image references (digests in production use), so a deploy + applies one set and rollback is re-deploying a previous pinned commit — no rolling + `:latest`/`:main`. _Until deployments are automated, the initial manual version keeps a + rolling `:main` reference; pinning to digests comes with the CI/CD work (see Next Steps)._ +2. **Two bundles, deployed independently.** `core` (sprue + signing-service + delegator + - hilt + plc + their dependency containers) and `piri` (the storage node + ingot S3 + gateway) deploy and roll back on their own cadence. They communicate over **public + `https://*.staging.fil.one` URLs** (fronted by the host's Caddy), not single-network + Docker DNS — so the split is real. +3. **Dependencies are per-environment.** Postgres + S3 (MinIO) + DynamoDB run as + containers in the stack; persistent data lives on the box's ZFS pool + (`/mnt/data/fil-one/forge`). The chain RPC is the **host Lotus node** (Calibnet, + `0.0.0.0:1234`), reached via `host.docker.internal:host-gateway` — no Anvil. +4. **Config in Git, secrets in 1Password.** Non-secret config (compose, `config.env`, + committed proofs) lives in the repo. Secret values live in the single 1Password item + `op://Fil One/FilOne Forge Staging` and are rendered onto the box at provision time — + never committed, never written to a developer's local disk, never in CI. + +There is **no** indexer / IPNI / redis / Anvil / mailer in staging. (sprue runs with an +empty `indexer.endpoint`; piri's base config omits the +`[ucan.services.indexer]`/`[publisher]` sections, which disables claim caching and IPNI +announcements; the delegator still validates indexing/egress delegations at startup, so +those proofs are generated even though neither service runs. ingot doesn't need an +indexer either — its reads resolve from its local `blob_locations` registry; see +[Running without an indexer](#running-without-an-indexer).) + +## Topology + +- **Box:** `root@23.83.66.244` (Servers.com Calibnet, hostname `ff`), Ubuntu, key-only SSH. Learn + more in [Servers.com Calibnet Box Runbook](https://www.notion.so/filecoin/Servers-com-Calibnet-Box-Runbook-36b7631f2825802b8e3ac9f25eadcc34#3907631f282580e198e2dfbc1e1a47ad) +- **Bundle `core`** — project `forge-staging-core` + - `sprue` (upload) + - `signing-service` + - `delegator` + - `hilt` (tenant management) + - `plc` (did:plc directory, **internal-only** — no public route) + - `hilt-vault` (persistent, sealed — Raft storage) + - `hilt-vault-unseal` (sidecar that unseals `hilt-vault` on every restart) + - `postgres` + - `minio` + - `dynamodb-local` +- **Bundle `piri`** — project `forge-staging-piri`. + - one `piri-0` storage node (postgres + filesystem) + - `ingot` (S3 gateway) + - `piri-postgres` +- **Postgres layout** + - one shared instance per bundle + - a dedicated `admin` superuser plus one role + database per service created by each bundle's `postgres-init` one-shot + - core: `sprue`, `hilt`, `plc` + - piri: `piri_0`, `ingot` +- The bundles talk over **public `https://*.staging.fil.one` URLs**, fronted by the host's existing Caddy (`caddy-guppy.service`). +- The host Lotus Eth RPC (`0.0.0.0:1234`) is reached from containers via `host.docker.internal:host-gateway`. +- **No** indexer / IPNI / redis / Anvil / smtp4dev in staging. +- **Logs** - Logs are automatically forwarded to our Grafana Cloud instance. + +Host layout: + +``` +/root/fil-one/forge/ # this repo, checked out on the box + environments/staging/smart-contracts.env # single source: chain id, RPC, contract addresses + environments/staging/wallets.env # wallet addresses (public), written by keygen + environments/staging/{core,piri}/... # bundle manifests + committed config + environments/staging/proofs/*.txt # committed UCAN proofs + caddy/forge-staging.caddy # copied from environments/staging/caddy/ +/root/fil-one/forge/secrets/ # provisioned (rendered) files (NOT in git) + delegator.pem + delegator.yaml + hilt.pem + ingot-config.yaml + ingot.pem + payer-key.hex + piri-0-wallet.hex + piri-0.pem + piri-base-config.toml + piri-secrets.env + secrets.env + signing-service.pem + sprue-config.yaml + sprue.pem + vault-secrets.env # hilt-vault unseal key + root token (shipped by staging-vault-init) +/mnt/data/fil-one/forge/ # persistent data on the ZFS pool + postgres/ # core bundle's shared Postgres (sprue, hilt, plc) + minio/ + dynamodb/ + hilt-vault/ # core bundle's Vault Raft store (survives restarts) + piri-0/ + piri-postgres/ # piri bundle's shared Postgres (piri_0, ingot) + ingot/ # ingot LSM segments, blob spool, token store +``` + +## Prerequisites + +- **DNS:** A records for `sprue` / `signing-service` / `delegator` / `piri-0` / `hilt` / `ingot` under + `staging.fil.one` → `23.83.66.244`, `proxied = false`. (Already set up via [infrastructure#35](https://github.com/fil-one/infrastructure/pull/35) and [infrastructure#37](https://github.com/fil-one/infrastructure/pull/37).) + + Deliberately **no `plc` record** — the did:plc directory is internal + to the core bundle (see [PLC is internal-only](#plc-is-internal-only)). + + No wildcard `*.ingot` record either — the S3 endpoint is path-style only (see + [S3 is path-style only](#s3-is-path-style-only)). + +- **Calibnet Forge contract addresses** — already filled in + [`environments/staging/smart-contracts.env`](../environments/staging/smart-contracts.env), + the **single source of truth** for the chain id, RPC URL, and every contract address. + Configs reference these as `${VAR}` and provision renders them in — no duplication, + nothing to fill by hand. Wallet addresses live in + [`environments/staging/wallets.env`](../environments/staging/wallets.env), written by keygen. +- **Dev machine:** + - `op` (1Password CLI, signed in) + - `ssh` to the box + - `go` (for keygen), + - `ucantool` (`go install github.com/fil-forge/ucantool@latest`) + - foundry's `cast` (for `staging-fund-payer`; install via ). + +## Initial deployment + +### 1. Ensure keys, wallets, proofs exist (idempotent) + +```bash +make staging-keygen # = go run ./cmd/smelt staging keygen +``` + +The ceremony has **ensure semantics** and is safe to re-run: every field the 1Password +item already holds is reused byte-for-byte (funded wallets, registered DIDs, and shipped +keys survive), and only missing fields are generated and added. Re-run it after pulling a +version that adds new services (e.g. hilt/ingot) — it mints only the new material and +reports which fields were reused vs generated. To rotate a specific secret, delete its +field from the 1Password item (and any proof files signed with it) and re-run. + +What's covered: + +- the Ed25519 identities (incl. `hilt` and `ingot`) +- three random EVM wallets (payer / delegator transactor / piri owner) +- connection secrets (Postgres admin + per-service passwords, MinIO keys, hilt partner key, ingot root S3 + credentials) + +The hilt Vault unseal key + root token are **not** covered here — Vault mints them at +runtime, so they are stored by [`make staging-vault-init`](#5a-initialize-the-vault-cross-cutting-step) +instead (see [Persistent sealed Vault](#persistent-sealed-vault)). + +Results: + +- stores all private material in the single 1Password item `op://Fil One/FilOne Forge Staging`, +- writes the UCAN proofs to `environments/staging/proofs/` (only missing ones, or those invalidated by a freshly generated key), +- writes the three public wallet addresses into `environments/staging/wallets.env` (`PAYER_ADDRESS` from there renders into piri's + config) +- prints the three EVM addresses + +### 2. Fund the wallets and commit + +Two token types are needed: + +- **tFIL (gas)** — fund all three addresses (in `wallets.env`) from the + [Calibnet faucet](https://faucet.calibnet.chainsafe-fil.io). +- **USDFC (storage payments)** — fund `PAYER_ADDRESS` from the + [Calibnet USDFC faucet](https://forest-explorer.chainsafe.dev/faucet/calibnet_usdfc). + This faucet caps out at **10 USDFC/day**. The USDFC must then be _deposited into the + FilecoinPay contract_ before piri can create a proof set — that's a separate on-chain + step, [`make staging-fund-payer`](#5b-fund-the-payers-filecoinpay-account), run around + deploy time (below). + +Then commit the generated artifacts: + +```bash +git add environments/staging/proofs environments/staging/wallets.env +git commit -m "staging: add delegation proofs and wallet addresses" +``` + +Contract addresses are already set in `smart-contracts.env`, so there is nothing else to +fill in. Keep `wallets.env` handy — top up both balances periodically. + +### 3. Bootstrap the box (first time only) + +Clones/updates the repo on the box, creates the secrets + data directories, wires the +Forge Caddy snippet into the host's main Caddyfile, opens UFW so containers can reach the +host's Caddy on `:443`, and verifies did:web endpoints (warn-only until the services are +deployed). Idempotent. + +```bash +make staging-bootstrap +``` + +(Override `FORGE_HOST`, `FORGE_DIR`, `MAIN_CADDYFILE`, `CADDY_SERVICE`, `REPO_URL`, `FORGE_REF`, +… as env vars; see `scripts/staging-bootstrap.sh`. `FORGE_REF` defaults to `main`; override it +to deploy an unmerged branch, e.g. `FORGE_REF=staging-deployment make staging-bootstrap`.) + +### 4. Provision secrets onto the box (dev machine) + +Provision the bundle you're about to deploy (we typically deploy one at a time): + +```bash +op signin +make staging-provision-core # sprue-config.yaml, delegator.yaml, secrets.env + key files (incl. hilt.pem) +make staging-provision-piri # piri-base-config.toml, ingot-config.yaml, piri-secrets.env + key files (incl. ingot.pem) +``` + +**Provisioning is destructive — it resets the bundle to a clean slate.** + +For each bundle it first removes the running containers and deletes that bundle's persistent data +dirs under `${FORGE_DATA_DIR}`, then recreates them empty. We wipe because Postgres bakes its +password into the data dir at first `initdb`, so a stale dir would keep an old password that no +longer matches the freshly provisioned secret. The next `staging-deploy` rebuilds everything from +scratch (sprue re-runs migrations, the delegator re-creates its DynamoDB tables, `minio-init` +re-creates buckets, piri re-syncs). + +For the **core** bundle the wipe also clears the `hilt-vault` Raft store, so the Vault must be +re-initialized: run [`make staging-vault-init`](#5a-initialize-the-vault-cross-cutting-step) after +provisioning core (see [Persistent sealed Vault](#persistent-sealed-vault)). + +This is safe because staging holds no precious data. + +After wiping it renders configs and ships key files (via `op read`) into +`/root/fil-one/forge/secrets/`, atomically, no plaintext on your local disk. + +Because the wipe is unconditional, re-running provision always discards existing +staging state. Run it on first deploy, after a `make staging-keygen` rotation, or +whenever you want a clean environment — not for an in-place config tweak you don't +want to lose data over. + +### 5. Deploy + +Initialize the Vault, deploy `core` first, allow-list the piri DID with the delegator, fund +the payer's FilecoinPay account, deploy `piri`, then run the two cross-bundle registrations: + +```bash +make staging-vault-init # init + unseal hilt-vault, store keys in 1Password (see 5a) +make staging-deploy-core # sprue + signing-service + delegator + hilt + plc + deps +make staging-allowlist-piri # add piri's DID to the delegator allow list +make staging-fund-payer # deposit USDFC into FilecoinPay (see 5b below) +make staging-deploy-piri # piri-0 + ingot +make staging-register-piri # register piri as a storage provider (see 6) +make staging-register-ingot # register ingot as hilt's regional provider (see 6b) +``` + +`staging-deploy-*` each sync the box's checkout to `FORGE_REF` (`git fetch` + +`reset --hard`), then pull the pinned images, recreate changed containers, and wait for +healthchecks (fails the deploy if any service stays unhealthy past the timeout). The deploy +**aborts** if the box has uncommitted changes to tracked files (untracked files such +as provisioned secrets are ignored). `FORGE_REF` defaults to `main`; override it to +deploy a branch, tag, or commit, e.g. `FORGE_REF=staging-deployment make staging-deploy-core`. + +**Why the allow-list step (order matters).** `piri init` step `[4/7]` ("Requesting +approval to join contract from Storacha") calls the delegator's +`/registrar/request-approval`, which **refuses any DID not on its allow list with a 403**. +In local dev, piri's entrypoint adds its own DID to the shared `dynamodb-local` allow list +before init; across the split bundles that route doesn't exist (the piri bundle can't reach +the core bundle's DynamoDB), so the staging entrypoint drops it. `staging-allowlist-piri` +fills the gap from the core side — it derives piri's DID from its provisioned key and runs +the delegator's `store allow-did` against the core DynamoDB. Run it **after** the delegator +is up (`deploy-core`) and **before** `deploy-piri`, so piri's first init is already +allow-listed. It is idempotent. Skipping it makes `deploy-piri` fail with piri crash-looping +on `registration failed with status: 403`. + +**`staging-provision-piri` must also have run first**: the allowlist/register scripts load +`$FORGE_SECRETS_DIR/piri-secrets.env` (shipped by provisioning) into their compose +invocations, and compose aborts on a missing env file — another reason §4 provisions both +bundles before this sequence starts. + +#### 5a. Initialize the Vault (cross-cutting step) + +`hilt-vault` is a real, sealed Vault (see [Persistent sealed Vault](#persistent-sealed-vault)), +so before deploying core it must be initialized once: + +```bash +make staging-vault-init +``` + +The script (`scripts/staging-vault-init.sh`, developer machine only) syncs the box checkout, +boots **only** the `hilt-vault` container (it needs no secrets), and then: + +- if Vault is **uninitialized** (fresh box, or after a re-provision wiped `/vault/file`) it + runs `vault operator init -key-shares=1 -key-threshold=1`, stores the resulting unseal key + + root token in the 1Password item (`hilt-vault-unseal-key`, `hilt-vault-root-token`), unseals + Vault, and enables KV v2 at `secret`; +- if Vault is **already initialized** it reuses the stored keys. + +Either way it renders `vault-secrets.env` from 1Password and ships it to the box. It is +idempotent, and secret values never touch local disk or a command line. + +**Why the order matters.** Unlike keygen's offline secrets, the unseal key and root token are +minted at _runtime_ by Vault, so they cannot exist until Vault runs. `staging-deploy-core` +consumes `vault-secrets.env` (the `hilt-vault-unseal` sidecar reads the unseal key from it and +hilt reads the root token), and its compose invocation **aborts on the missing env file** if +`staging-vault-init` hasn't run. Run it **after** `staging-provision-core` (which wipes +`/vault/file`) and **before** `staging-deploy-core`. + +#### 5b. Fund the payer's FilecoinPay account + +`piri init` step `[4/6]` ("Setting up proof set") asks FilecoinPay to lock up a fixed +amount (~0.9 USDFC) on the payer's behalf. **Lockup can only draw on funds deposited _into_ +the FilecoinPay contract — not the USDFC sitting in the payer wallet.** So a wallet that the +USDFC faucet just topped up still fails with: + +``` +Error: creating proof set: failed to send transaction: +InsufficientLockupFunds(Payer=0x…, MinimumRequired=900000000000000000, Available=0) +``` + +The local dev stack never hits this: its Anvil baseline ships with the deposit and operator +approval baked into the chain state. On Calibnet nobody seeded that, and `piri init` does not +deposit for you (its `setupProofSet` calls `CreateProofSet` directly) — so we do it once, out +of band: + +```bash +make staging-fund-payer +``` + +The script (`scripts/staging-fund-payer.sh`, developer machine only) reads the payer key from +1Password and sends three transactions to Calibnet with `cast`: + +1. `USDFC.approve(FilecoinPay, amount)` — let FilecoinPay pull the USDFC. +2. `FilecoinPay.deposit(USDFC, payer, amount)` — credit the payer's FilecoinPay account (this + is what clears `Available=0`). +3. `FilecoinPay.setOperatorApproval(USDFC, FWSS, …)` — let the warm-storage service lock up + funds when it creates the proof-set rail (missing this trades the funds error for an + approval revert). + +Contract addresses come from `smart-contracts.env`, the payer from `wallets.env`. Amounts are +baked in but overridable via env vars (`USDFC_DEPOSIT_AMOUNT`, `USDFC_LOCKUP_ALLOWANCE`, +`USDFC_RATE_ALLOWANCE`, `USDFC_MAX_LOCKUP_PERIOD`); defaults are kept **well below 5 USDFC** so +one daily faucet grant (10 USDFC/day) covers a top-up with headroom. It reads current balances +first and **skips the deposit if the account already holds enough** (override with +`FORCE_DEPOSIT=1`), so re-running is safe. The RPC defaults to the public glif Calibnet +endpoint (`CALIBNET_RPC_URL` to override — the box's `host.docker.internal` Lotus URL in +`smart-contracts.env` is container-only and not reachable from your dev machine). + +Run it **after** `deploy-core` (nothing here depends on it, but the payer wallet must already +hold USDFC) and **before** `deploy-piri`, so piri's first init finds the lockup funds +available. Like the wallet balances, this is a **periodic top-up chore** — re-run it if +proof-set operations later fail with `InsufficientLockupFunds`. + +### 6. Register the piri provider with the core (cross-bundle step) + +In local dev, sprue's `post_start.sh` auto-registers piri providers; across bundles that +can't run, so register explicitly once both bundles are healthy. Without this step, +uploads fail with `CandidateUnavailable: no storage providers available`. + +```bash +make staging-register-piri +``` + +The script (`scripts/staging-register-piri.sh`) derives the piri DID from the piri +bundle's provisioned key, then runs the same admin calls as the local post_start hook +inside the sprue container: `sprue client admin provider register +https://piri-0.staging.fil.one /proofs/piri-0-proof.txt` followed by `provider weight set + 100 100`. The committed proofs directory is mounted into sprue at `/proofs` by the +core compose file. Idempotent: an already-registered provider is tolerated and the weight +is simply re-applied. + +#### 6b. Register ingot with hilt (cross-bundle step) + +In local dev, hilt's `post_start.sh` registers ingot as the regional S3 provider; across +bundles that can't run either. Without this step hilt rejects tenant creation for the +region and every `/s3/*` invocation from ingot. + +```bash +make staging-register-ingot +``` + +The script (`scripts/staging-register-ingot.sh`) derives ingot's did:key via `ingot +whoami` in the piri bundle, then runs `hilt client admin provider add eu-central-3` +inside the hilt container (region overridable via `INGOT_REGION` — it must match the +`region` in `ingot-config.yaml` and the `AWS_REGION` S3 clients sign with). Run it after +both bundles are healthy and **before creating any tenants**. Idempotent. + +This is the **only** cross-bundle registration hilt needs: its authority to call sprue's +`/customer/add` comes entirely from the committed `hilt-customer-add-proof.txt`, and sprue +resolves `did:web:hilt.staging.fil.one` over https at invocation time. + +### 7. Verify + +```bash +# Health +ssh root@23.83.66.244 'cd /root/fil-one/forge/environments/staging/core && docker compose -p forge-staging-core ps' +ssh root@23.83.66.244 'cd /root/fil-one/forge/environments/staging/piri && docker compose -p forge-staging-piri ps' + +# did:web resolution through Caddy +# Only sprue, delegator, and hilt serve a did.json. signing-service does NOT — it has +# no /.well-known/did.json route by design (it resolves its own DID from an in-memory +# document, and no peer ever resolves did:web:signing-service: piri uses the +# configured DID only as the signing-invocation audience, and the signed response +# is an EIP-712 signature verified on-chain, not a did:web-resolved UCAN receipt). +# So a 404 there is expected, not a failure. ingot serves no did.json either — it +# acts under a did:key. +for h in sprue delegator hilt; do curl -fsS "https://$h.staging.fil.one/.well-known/did.json" >/dev/null && echo "$h ok"; done + +# ingot (S3 gateway) health +curl -fsS https://ingot.staging.fil.one/health && echo "ingot ok" +``` + +## Post-deploy + +### End-to-end smoke test (Hilt/Ingot S3 flow) + +The target architecture is deployed: FilOne calls hilt's Tenant API to register tenants +and issue S3 access keys; the data plane uses standard S3 primitives against ingot. The +flow below exercises it end-to-end: tenant and access-key creation, bucket operations, and +the S3 object write/read roundtrip. + +Run after both bundles are healthy and both registrations are done ([§6](#6-register-the-piri-provider-with-the-core-cross-bundle-step), +[§6b](#6b-register-ingot-with-hilt-cross-bundle-step)). Everything runs from your dev +machine against the public URLs. + +**1. Create a tenant and an access key** (Tenant API, bearer = the partner key): + +```bash +PARTNER_KEY=$(op read "op://Fil One/FilOne Forge Staging/hilt-partner-key") + +# Create a tenant in the region ingot is registered under (eu-central-3). +curl -si -X PUT "https://hilt.staging.fil.one/tenants/smoke-1" \ + -H "Authorization: Bearer $PARTNER_KEY" -H "Content-Type: application/json" \ + -d '{"region":"eu-central-3"}' +# → 201; hilt publishes the tenant did:plc (internal plc) and registers it as a +# customer with sprue via /customer/add (watch sprue's logs to confirm). + +# Mint an S3 access key for the tenant. `name` and at least one permission are +# required; valid permissions are the AWS-style strings in hilt's +# pkg/s3perm/s3perm.go (s3:GetObject, s3:PutObject, s3:CreateBucket, ...). +# The secret is returned ONCE, so pipe it straight into a .env file that sets +# all four AWS_ vars step 2 needs; `source .env.staging` before running the S3 test. +# --fail-with-body makes curl exit non-zero on a non-2xx status (still printing +# the error body). Capture first so a failure doesn't clobber .env.staging via >. +if resp=$(curl -sS --fail-with-body -X POST "https://hilt.staging.fil.one/tenants/smoke-1/access-keys" \ + -H "Authorization: Bearer $PARTNER_KEY" -H "Content-Type: application/json" \ + -d '{"name":"smoke","permissions":["s3:CreateBucket","s3:ListAllMyBuckets","s3:DeleteBucket","s3:ListBucket","s3:ListBucketVersions","s3:GetObject","s3:PutObject","s3:DeleteObject","s3:GetObjectRetention","s3:PutObjectRetention","s3:GetObjectLegalHold","s3:PutObjectLegalHold","s3:GetObjectVersion","s3:DeleteObjectVersion"]}'); then + jq -r ' + "AWS_ACCESS_KEY_ID=" + .accessKeyId, + "AWS_SECRET_ACCESS_KEY=" + .secretAccessKey, + "AWS_REGION=eu-central-3", + "AWS_ENDPOINT_URL=https://ingot.staging.fil.one" + ' <<<"$resp" > .env.staging + echo "credentials for access key $(jq -r .accessKeyId <<<"$resp") saved to .env.staging" +else + echo "access-key mint failed:" >&2; echo "$resp" >&2 +fi +``` + +**2. S3 against ingot.** + +Two client-side settings matter: + +- **Use region `eu-central-3`** (set in `AWS_REGION`) +- **Path-style addressing** (no wildcard `*.ingot` DNS/TLS exists). The aws CLI + has no env var for this — it's a config-file setting. Run this once: + + ```bash + aws configure set default.s3.addressing_style path + ``` + +Test script: + +```bash +set -a; source .env.staging; set +a # AWS_ACCESS_KEY_ID/SECRET/REGION/ENDPOINT_URL from step 1 + +aws s3 mb s3://smoke-bucket +head -c 10240 /tmp/hello.bin +aws s3 cp /tmp/hello.bin s3://smoke-bucket/hello.bin +aws s3 cp s3://smoke-bucket/hello.bin /tmp/out.bin +cmp /tmp/hello.bin /tmp/out.bin && echo "roundtrip ok" +``` + +Behind the scenes: ingot authorizes each request against hilt (`/s3/request/authorize`, +authority = the committed `hilt-ingot-s3-proof.txt`), spools the object, and ships CAR +segments to sprue → piri over public https. Reads resolve from ingot's local +`blob_locations` registry — no indexer involved (see +[Running without an indexer](#running-without-an-indexer)). + +### Updating a bundle after a new image is published + +Once a bundle is up, picking up a freshly-published service image (e.g. a new +`hilt`/`plc` in core, or `ingot`/`piri` in piri) is just a **redeploy** of that bundle: + +```bash +make staging-deploy-core # refreshes core images (hilt, plc, sprue, ...) +make staging-deploy-piri # refreshes piri images (piri-0, ingot) +``` + +Because `versions.env` still tracks the rolling `:main` tag, `staging-deploy-*` runs +`docker compose pull` (fetching the new image under the same tag) and then recreates only +the containers whose image digest changed — unchanged services keep running. Data survives: +a redeploy is **not** provisioning, so it does not touch secrets or wipe the ZFS-backed +volumes. + +Notes: + +- **Do not re-run `make staging-provision-*`** to pick up an image — provisioning is + destructive (§4) and would wipe the bundle's data. +- **No cross-bundle re-registration is needed** for a plain image bump. The allow-list, + payer funding, and provider registrations (§§5b, 6, 6b) persist across a redeploy; only + re-run them if provisioning wiped state or a DID/key changed. +- If a service's **config template** changed with the new image (not just the binary), + re-provision that bundle so the rendered config is regenerated — for the piri base config + see the re-provision-then-deploy caveat under + [Running without an indexer](#running-without-an-indexer). +- Once images are pinned to `@sha256:` digests (see [Improvements](#improvements)), updating + instead means bumping the digest in `versions.env`, committing, and redeploying — the pull + becomes a no-op and the new digest drives the recreate. + +### Rollback + +_This will be applicable in the future, after we have pinned versions._ + +Re-deploy a previous pinned commit (deploy checks it out on the box for you): + +```bash +FORGE_REF= make staging-deploy-core +FORGE_REF= make staging-deploy-piri +``` + +Image versions are pinned per bundle in `versions.env`, so rolling back the application +versions is deterministic. Data/schema rollback is out of scope. + +### One-shot reset + +To wipe **all** staging data and redeploy both bundles from scratch — the full §4 +(provision both bundles) + §5 (vault-init, deploy, fund) + §6 (register) sequence in the +correct order — run: + +```bash +# Using the current `main` branch +make staging-reset + +# Using a specific branch/commit/tag (e.g. a PR branch) +FORGE_REF=staging-deployment make staging-reset +``` + +`staging-reset` chains the nine `staging-*` targets via recursive `make` and **aborts on +the first failure** (every step is idempotent, so re-running from the top after a transient +failure — e.g. a flaky Calibnet RPC in `staging-fund-payer` — is safe). It is **destructive**: +it discards Hilt tenants + access keys, piri objects, ingot buckets + `blob_locations`, +Postgres, DynamoDB, MinIO, and the `hilt-vault` Raft store. + +**What survives:** all Ed25519 service identities and EVM wallet addresses (only re-shipped +from 1Password, never regenerated — so no re-funding or `wallets.env` re-commit) and all +on-chain state (wallet balances, the payer's FilecoinPay deposit). **What rotates:** the +`hilt-vault` unseal key + root token, since the wiped Vault is re-initialized by +`staging-vault-init` (which overwrites those two 1Password fields — see +[Persistent sealed Vault](#persistent-sealed-vault)). + +After it finishes, re-create tenants and buckets via the Tenant API / S3 flow ([End-to-end smoke test](#end-to-end-smoke-test-hiltingot-s3-flow)). + +### Topping up low wallet balances + +The staging stack draws on three funded wallets, and their balances deplete over time — +gas is spent on every on-chain transaction, and USDFC is consumed as storage payments +settle. When a balance runs dry the stack fails silently: `piri init` and later proof-set +operations break with `InsufficientLockupFunds`, and uploads stall. Until balance +monitoring/alerting exists (Major Gap #4), check and top up periodically — and whenever an +on-chain step fails unexpectedly. + +The addresses are in [`environments/staging/wallets.env`](../environments/staging/wallets.env) +(`PAYER_ADDRESS`, delegator transactor, piri owner). There are two balances to watch: + +**1. tFIL (gas) — all three wallets.** Every wallet needs tFIL to send transactions. +Top up from the [Calibnet faucet](https://faucet.calibnet.chainsafe-fil.io) (§2). Check a +balance with `cast`: + +```bash +cast balance
--rpc-url https://api.calibration.node.glif.io/rpc/v1 --ether +``` + +**2. USDFC / FilecoinPay lockup — payer only.** Storage payments draw on USDFC that has +been _deposited into the FilecoinPay contract_, not the raw balance in the payer wallet. +Two-step top-up: + +- Fund `PAYER_ADDRESS` with USDFC from the + [Calibnet USDFC faucet](https://forest-explorer.chainsafe.dev/faucet/calibnet_usdfc) + (caps at 10 USDFC/day). +- Deposit it into FilecoinPay with [`make staging-fund-payer`](#5b-fund-the-payers-filecoinpay-account), + which reads current balances and skips the deposit if the account already holds enough + (override with `FORCE_DEPOSIT=1`). Re-run it whenever proof-set operations fail with + `InsufficientLockupFunds`. + +See [§2](#2-fund-the-wallets-and-commit) and [§5b](#5b-fund-the-payers-filecoinpay-account) +for the full first-time funding procedure. + +## Configuration notes + +Deliberate, verified aspects of how the staging stack is configured. + +### Running without an indexer + +Staging deliberately runs no indexing-service or IPNI (see +[Architecture](#architecture)). The rest of the stack is +configured to operate cleanly without them, and this is confirmed working: + +- **sprue** takes an empty `indexer.endpoint`/`did`, which disables its indexer calls. +- **delegator** still validates the indexing + egress-tracking service DIDs and their proofs + at startup, so keygen issues those proofs and `delegator.yaml` references them even though + neither service runs. +- **piri** omits the `[ucan.services.indexer]` and `[publisher]` sections from its base + config, which turns off claim caching and IPNI announcements. +- **ingot** resolves blob locations from its own `blob_locations` Postgres table (the + appliance read tier) rather than an indexing-service, so the missing indexer doesn't + affect the S3 read path. + +Because `piri init` bakes the base config into a node's merged config, changing these +sections requires re-provisioning and re-deploying the piri bundle before a node picks them +up. + +### Self-initializing state (tables, buckets, migrations) + +A freshly-provisioned bundle initializes its own persistent state on first deploy — no +manual setup is needed: + +- the **delegator** auto-creates its DynamoDB tables (`delegator-allow-list`, + `delegator-provider-info`), +- **minio-init** creates sprue's buckets, +- **sprue** runs its own Postgres migrations. + +Each bundle's postgres healthcheck probes `127.0.0.1` (`pg_isready -h 127.0.0.1`), +not the default unix socket. On a freshly-wiped data dir the postgres image runs a +temporary bootstrap server on the unix socket only (`listen_addresses=''`) while it +runs `initdb`; a socket-based `pg_isready` would report healthy during that window, +so `postgres-init` — which connects over TCP — could start too early and fail with +`psql` exit 2, failing the deploy. The TCP probe stays unready until the real server +is listening on TCP, closing the race. + +### Payment-plan bypass + +sprue runs with `deployment.allow_provision_without_payment_plan: true` (in +`environments/staging/core/config/sprue/config.yaml.tpl`), so spaces can be provisioned +without a customer payment plan. Storacha payment plans are a left-over sprue inherited; +FilOne Forge uses a different billing mechanism and does not need them, so we bypass the +check here. + +### Persistent sealed Vault + +`hilt-vault` runs HashiCorp Vault with the **integrated Raft** storage backend, persisted +on the ZFS pool at `/mnt/data/fil-one/forge/hilt-vault/`. It is a **real, sealed** Vault — +not dev mode — so tenant/access-key private keys **survive restarts** (deploy recreate, box +reboot, `docker restart hilt-vault`). + +Two pieces make that work: + +- **`hilt-vault-unseal` sidecar** — a long-running poller (`restart: unless-stopped`) that + unseals `hilt-vault` whenever it is found sealed, i.e. on every (re)start. It reads the + single unseal key from `vault-secrets.env`. Vault's `/v1/sys/health` stays unhealthy + while sealed, so hilt's `depends_on hilt-vault: service_healthy` waits until the sidecar + has unsealed it. +- **[`make staging-vault-init`](#5a-initialize-the-vault-cross-cutting-step)** — the + one-time (idempotent) ceremony that runs `vault operator init` against a fresh Vault, + enables KV v2 at `secret`, stores the unseal key + root token in the 1Password item, and + ships `vault-secrets.env` to the box. Unlike every other secret, these two values are + minted at _runtime_ by Vault (not offline by keygen), which is why they have their own + step and their own env file. + +The Vault root token doubles as hilt's client token (`HILT_VAULT_TOKEN`); a scoped +policy + limited token is a future hardening step ([Next Steps](#next-steps)). Hilt supports +only `memory` and `hashicorp` vault backends. + +**A re-provision (`make staging-provision-core`) wipes `/vault/file`**, so it discards the +initialized Vault (and thus every tenant access key). Re-run `make staging-vault-init` after +any provision — it re-initializes the empty Vault and overwrites the two 1Password fields. +Hilt's Postgres rows survive a Vault wipe but then reference vault entries that no longer +exist, so re-create affected tenants via the Tenant API. This is acceptable because staging +holds no precious data. + +### PLC is internal-only + +The did:plc directory runs inside the core bundle with no published port, no Caddy route, +and no DNS record. Its only consumers are hilt (`HILT_PLC_DIRECTORY=http://plc:3000`) and +sprue (`deployment.plc_directory`), both in-bundle. Consequence: tenant `did:plc` +identities are **not publicly resolvable** from outside the box. Add a Caddy route + DNS +record later if external resolution is ever needed. + +### S3 is path-style only + +There is no wildcard `*.ingot.staging.fil.one` DNS record or certificate, so +virtual-hosted bucket addressing (`bucket.ingot.staging.fil.one`) does not work. Every S3 +client must force path-style (`aws configure set default.s3.addressing_style path`, +`forcePathStyle: true`, etc.). + +### Provisioning wipes bundle data + +`make staging-provision-*` deliberately resets a bundle to a clean slate (see +[§4](#4-provision-secrets-onto-the-box-dev-machine)). For the piri bundle the wipe covers +`piri-0`, `piri-postgres`, and `ingot`: buckets, objects, and ingot's `blob_locations` +registry all vanish, orphaning the read-side location knowledge of previously shipped +objects. Hilt tenants (core bundle) survive, but their buckets' data-plane state is gone — +re-create buckets after a piri re-provision. + +## Next Steps + +### Major Gaps + +1. Write an automated script for the Hilt/Ingot end-to-end workflow ([End-to-end smoke test](#end-to-end-smoke-test-hiltingot-s3-flow)). +2. Monitoring & alerting for wallet balances. The three wallets' tFIL (gas) and the + payer's USDFC / FilecoinPay lockup balances are manual top-up chores today (§2, + [§5b](#5b-fund-the-payers-filecoinpay-account)) with nothing watching them — when + any runs dry the stack fails silently (piri init / proof-set ops break with + `InsufficientLockupFunds`, uploads stall). Add a periodic balance check that alerts + (Grafana Cloud, where logs already ship) before a wallet crosses a low-balance + threshold, and a runbook entry for topping up — see + [Topping up low wallet balances](#topping-up-low-wallet-balances). +3. Harden the deploy health gate against recovered-crash false positives. It fails any + container with `RestartCount > 0` even after the container recovered and is healthy + (the counter persists for the container's lifetime), so `staging-deploy-*` can report + `crash-looping (restarts=N)` for a service `docker compose ps` shows healthy. Until it + is fixed, clear the counter by recreating just that container (`docker compose -p + up -d --force-recreate `) and re-run the deploy — after + checking `docker logs` for the original crash. + +### Continuous Deployment + +1. Pin every application image in `versions.env` to a `@sha256:` digest (currently rolling + `:main` placeholders) before a real deploy — resolve digests with `docker buildx +imagetools inspect …`; this covers `HILT_IMAGE`, `PLC_IMAGE` (core) and `INGOT_IMAGE` + (piri) too. +2. Implement GH Action workflows to automatically upgrade pinned versions whenever a new Docker + image version is pushed by each service repository (sprue, piri, etc). +3. Implement a CI/CD workflow to automatically deploy every commit landed to `main` to the staging box. +4. Add a basic end-to-end test suite (smoke tests). + +The outcome of the above: every commit landed to `main` in any Forge repository is automatically +deployed to the staging box (after it passes CI checks in the original repository and the e2e smoke +tests in Smelt). + +Before we invest more into improving the Docker Compose setup, we should have a discussion about how +we want to run Forge in production. The current Compose setup has many short-comings, e.g. deploys are +not atomic and there is no support for rolling upgrades or blue/green deployments. + +The staging box is already running [K3s](https://k3s.io/), we could consider moving the core bundle +to a Kubernetes deployment + +### Improvements + +1. Give hilt a scoped Vault policy + limited token instead of the root token. + `hilt-vault` is now persistent and sealed (see + [Persistent sealed Vault](#persistent-sealed-vault)), but hilt currently + authenticates with the init-generated **root** token. `staging-vault-init` + should instead write a KV-scoped policy and mint a limited token for hilt, + keeping the root token 1Password-only for admin use. diff --git a/environments/staging/README.md b/environments/staging/README.md new file mode 100644 index 0000000..89f2b00 --- /dev/null +++ b/environments/staging/README.md @@ -0,0 +1,16 @@ +# Forge staging deployment + +Version-pinned Docker Compose manifests and configuration for the Forge **staging** +environment on the Servers.com Calibnet box (`root@23.83.66.244`). The stack is split +into two independently-deployed bundles — [`core/`](core/) (sprue + signing-service + +delegator + hilt tenant management + the internal-only plc did:plc directory + +hilt-vault and their dependency containers) and [`piri/`](piri/) (a piri storage node + +the ingot S3 gateway and their shared Postgres) — which talk to each other over public +`https://*.staging.fil.one` URLs fronted by the host's Caddy reverse proxy. Image +versions are pinned in each bundle's `versions.env`; non-secret config lives in +committed `config.env` / templates, and secret values come from 1Password at provision +time (never committed). Delegation proofs in [`proofs/`](proofs/) are generated by +`smelt staging keygen` (idempotent — safe to re-run) and committed. + +See **[docs/STAGING_DEPLOY.md](../../docs/STAGING_DEPLOY.md)** for the full fresh-host +bootstrap and deploy runbook. diff --git a/environments/staging/caddy/forge-staging.caddy b/environments/staging/caddy/forge-staging.caddy new file mode 100644 index 0000000..6ee929d --- /dev/null +++ b/environments/staging/caddy/forge-staging.caddy @@ -0,0 +1,44 @@ +# Forge staging — Caddy site blocks. +# +# Kept in the repo and shipped to the box under ~/fil-one/forge/caddy/. Import it +# from the existing main Caddyfile (~/storacha/caddy/Caddyfile) with one line: +# +# import /root/fil-one/forge/caddy/*.caddy +# +# then reload: systemctl reload caddy-guppy +# +# Each host terminates TLS (automatic ACME) and reverse-proxies to the bundle's +# loopback-published container port. This also serves did:web resolution: +# https:///.well-known/did.json is proxied straight from the service. +# +# Prerequisite: A records for these hosts must point at the box (23.83.66.244), +# proxied=false — see ../dns/fil-one-staging.tf. + +sprue.staging.fil.one { + reverse_proxy 127.0.0.1:15060 +} + +signing-service.staging.fil.one { + reverse_proxy 127.0.0.1:15030 +} + +delegator.staging.fil.one { + reverse_proxy 127.0.0.1:15040 +} + +piri-0.staging.fil.one { + reverse_proxy 127.0.0.1:15100 +} + +hilt.staging.fil.one { + reverse_proxy 127.0.0.1:15110 +} + +# S3 endpoint (ingot). PATH-STYLE ONLY: no wildcard *.ingot.staging.fil.one +# DNS/TLS exists, so virtual-hosted bucket addressing does not work — clients +# must set force_path_style / addressing_style=path. Caddy imposes no default +# body-size limit and streams proxied bodies, so large S3 PUTs need no extra +# directives. ingot serves no did.json (it acts under a did:key). +ingot.staging.fil.one { + reverse_proxy 127.0.0.1:15130 +} diff --git a/environments/staging/core/compose.yml b/environments/staging/core/compose.yml new file mode 100644 index 0000000..88abb93 --- /dev/null +++ b/environments/staging/core/compose.yml @@ -0,0 +1,407 @@ +# Forge staging — CORE bundle. +# +# upload (sprue) + signing-service + delegator + hilt (tenant management) + plc +# (did:plc directory) + hilt-vault, plus their dependency containers (postgres, +# minio, dynamodb-local). Deployed independently of the piri bundle; the two talk +# over public https://*.staging.fil.one URLs (fronted by host Caddy). +# +# Postgres is a single shared instance: a dedicated `admin` superuser plus one +# role + database per service (sprue, hilt, plc), created by the postgres-init +# one-shot. +# +# Run via the deploy script, which supplies the env files: +# docker compose -p forge-staging-core \ +# --env-file versions.env --env-file config.env --env-file secrets.env \ +# up -d +# +# All published ports bind to 127.0.0.1 (the box's UFW default-denies inbound and +# Docker's iptables can bypass it — loopback binding keeps these off the public +# interface; Caddy fronts them for TLS + did:web). +name: forge-staging-core + +services: + dynamodb-local: + image: ${DYNAMODB_IMAGE} + user: "0:0" + command: ["-jar", "DynamoDBLocal.jar", "-sharedDb", "-dbPath", "/data"] + volumes: + - ${FORGE_DATA_DIR}/dynamodb:/data + healthcheck: + test: ["CMD-SHELL", "bash -c ':>/dev/tcp/127.0.0.1/8000'"] + start_interval: 1s + interval: 5s + timeout: 3s + retries: 5 + start_period: 10s + restart: unless-stopped + networks: [forge-staging] + + postgres: + image: ${POSTGRES_IMAGE} + # Shared instance with a dedicated admin superuser. The per-service roles + + # databases (sprue, hilt, plc) are created by postgres-init below; each + # service runs its own migrations in its own database. + environment: + - POSTGRES_USER=admin + - POSTGRES_PASSWORD=${POSTGRES_ADMIN_PASSWORD} + volumes: + - ${FORGE_DATA_DIR}/postgres:/var/lib/postgresql/data + healthcheck: + # -h 127.0.0.1 forces a TCP check. On a freshly-wiped data dir the image + # runs a temporary bootstrap server (listen_addresses='') on the unix + # socket only while it runs initdb; a socket-based pg_isready reports + # healthy during that window, so postgres-init connects over TCP too early + # and psql exits 2. A TCP probe stays unready until the real server listens. + test: ["CMD-SHELL", "pg_isready -h 127.0.0.1 -U admin -d admin"] + start_interval: 1s + interval: 5s + timeout: 3s + retries: 10 + start_period: 10s + restart: unless-stopped + networks: [forge-staging] + + # Create the per-service Postgres roles + databases once postgres is up. + # Idempotent: guarded CREATEs (via \gexec — CREATE DATABASE cannot run inside + # a DO block/transaction) plus an unconditional ALTER ROLE so a re-provisioned + # password always applies. Exits 0; consumers wait on completion. Passwords + # land inside SQL string literals — safe only because keygen secrets are hex + # ([0-9a-f]), never quotes or backslashes. + postgres-init: + image: ${POSTGRES_IMAGE} + depends_on: + postgres: + condition: service_healthy + entrypoint: + - /bin/sh + - -c + - | + set -eu + export PGPASSWORD="$$POSTGRES_ADMIN_PASSWORD" + ensure() { + printf '%s\n' \ + "SELECT 'CREATE ROLE \"$$1\"' WHERE NOT EXISTS (SELECT FROM pg_roles WHERE rolname = '$$1')" \ + '\gexec' \ + "ALTER ROLE \"$$1\" WITH LOGIN PASSWORD '$$2';" \ + "SELECT 'CREATE DATABASE \"$$1\" OWNER \"$$1\"' WHERE NOT EXISTS (SELECT FROM pg_database WHERE datname = '$$1')" \ + '\gexec' \ + | psql -h postgres -U admin -d admin -v ON_ERROR_STOP=1 --quiet + } + ensure sprue "$$SPRUE_POSTGRES_PASSWORD" && + ensure hilt "$$HILT_POSTGRES_PASSWORD" && + ensure plc "$$PLC_POSTGRES_PASSWORD" && + echo "postgres roles + databases ready" + environment: + POSTGRES_ADMIN_PASSWORD: ${POSTGRES_ADMIN_PASSWORD} + SPRUE_POSTGRES_PASSWORD: ${SPRUE_POSTGRES_PASSWORD} + HILT_POSTGRES_PASSWORD: ${HILT_POSTGRES_PASSWORD} + PLC_POSTGRES_PASSWORD: ${PLC_POSTGRES_PASSWORD} + restart: "no" + networks: [forge-staging] + + minio: + image: ${MINIO_IMAGE} + command: server /data --console-address ":9001" + environment: + MINIO_ROOT_USER: ${MINIO_ROOT_USER} + MINIO_ROOT_PASSWORD: ${MINIO_ROOT_PASSWORD} + ports: + - "127.0.0.1:15072:9001" # console (optional, loopback only) + volumes: + - ${FORGE_DATA_DIR}/minio:/data + healthcheck: + test: ["CMD", "mc", "ready", "local"] + start_interval: 1s + interval: 5s + timeout: 3s + retries: 5 + start_period: 10s + restart: unless-stopped + networks: [forge-staging] + + # Create sprue's buckets once minio is up. Exits 0; sprue waits on completion. + minio-init: + image: ${MC_IMAGE} + depends_on: + minio: + condition: service_healthy + entrypoint: + - /bin/sh + - -c + - | + mc alias set local http://minio:9000 "$$MINIO_ROOT_USER" "$$MINIO_ROOT_PASSWORD" && + mc mb -p local/agent-message local/delegation local/upload-shards && + echo "buckets ready" + environment: + MINIO_ROOT_USER: ${MINIO_ROOT_USER} + MINIO_ROOT_PASSWORD: ${MINIO_ROOT_PASSWORD} + restart: "no" + networks: [forge-staging] + + signing-service: + image: ${SIGNER_IMAGE} + user: root + ports: + - "127.0.0.1:15030:7446" + extra_hosts: + - "host.docker.internal:host-gateway" + volumes: + - ${FORGE_SECRETS_DIR}/payer-key.hex:/keys/payer-key.hex:ro + - ${FORGE_SECRETS_DIR}/signing-service.pem:/keys/signing-service.pem:ro + # signer.yaml carries no secrets and is committed; mounted from the git tree. + - ./config/signer/signer.yaml:/signer.yaml:ro + command: + - "--host" + - "0.0.0.0" + - "--port" + - "7446" + - "--rpc-url" + - "${LOTUS_RPC_URL}" + - "--service-contract-address" + - "${FWSS_ADDRESS}" + - "--signing-key-path" + - "/keys/payer-key.hex" + - "--service-key-file" + - "/keys/signing-service.pem" + - "--service-did" + - "${SIGNING_SERVICE_DID}" + healthcheck: + test: ["CMD", "wget", "-q", "--spider", "http://localhost:7446/healthcheck"] + start_interval: 1s + interval: 10s + timeout: 5s + retries: 5 + start_period: 10s + restart: unless-stopped + networks: [forge-staging] + + delegator: + image: ${DELEGATOR_IMAGE} + user: "0:0" + ports: + - "127.0.0.1:15040:80" + command: ["serve", "--host", "0.0.0.0", "--port", "80"] + extra_hosts: + - "host.docker.internal:host-gateway" + volumes: + - ${FORGE_SECRETS_DIR}/delegator.pem:/keys/delegator.pem:ro + - ${FORGE_SECRETS_DIR}/delegator.yaml:/.delegator.yaml:ro + # Committed delegation proofs (non-secret) come from the git tree. + - ../proofs:/proofs:ro + depends_on: + dynamodb-local: + condition: service_healthy + healthcheck: + test: ["CMD", "wget", "-q", "--spider", "http://localhost:80/healthcheck"] + start_interval: 1s + interval: 10s + timeout: 5s + retries: 5 + start_period: 10s + restart: unless-stopped + networks: [forge-staging] + + sprue: + image: ${SPRUE_IMAGE} + command: ["serve", "--config", "/etc/sprue/config.yaml"] + ports: + - "127.0.0.1:15060:80" + extra_hosts: + - "host.docker.internal:host-gateway" + environment: + # S3/MinIO credentials are read from the mounted sprue-config.yaml + # (storage.s3.access_key_id / secret_access_key), not from AWS_* env vars. + - SPRUE_SERVER_PUBLIC_URL=${SPRUE_PUBLIC_URL} + volumes: + - ${FORGE_SECRETS_DIR}/sprue.pem:/keys/sprue.pem:ro + - ${FORGE_SECRETS_DIR}/sprue-config.yaml:/etc/sprue/config.yaml:ro + # Committed delegation proofs (non-secret) — provider registration + # (scripts/staging-register-piri.sh) passes /proofs/piri-N-proof.txt. + - ../proofs:/proofs:ro + # NOTE: unlike local dev, there is NO post_start provider-registration hook — + # piri lives in a separate bundle. Register piri providers explicitly after + # both bundles are healthy: `make staging-register-piri`. + depends_on: + postgres: + condition: service_healthy + postgres-init: + condition: service_completed_successfully + minio: + condition: service_healthy + minio-init: + condition: service_completed_successfully + healthcheck: + test: ["CMD", "curl", "-sf", "http://localhost:80/health"] + start_interval: 1s + interval: 10s + timeout: 5s + retries: 5 + start_period: 10s + restart: unless-stopped + networks: [forge-staging] + + # did:plc directory (tenant DIDs). INTERNAL ONLY by design: no published port, + # no Caddy route, no DNS record — its only consumers are hilt and sprue inside + # this bundle (http://plc:3000), so did:plc documents are NOT publicly + # resolvable in staging. + plc: + image: ${PLC_IMAGE} + environment: + # The production entrypoint takes DB credentials as JSON, with a separate + # (here: identical) credentialed user for startup migrations. The password + # lands inside a JSON value — safe only because keygen secrets are hex + # ([0-9a-f]), never quotes or backslashes. + - DB_CREDS_JSON={"username":"plc","password":"${PLC_POSTGRES_PASSWORD}","host":"postgres","port":"5432","database":"plc"} + - DB_MIGRATE_CREDS_JSON={"username":"plc","password":"${PLC_POSTGRES_PASSWORD}","host":"postgres","port":"5432","database":"plc"} + - ENABLE_MIGRATIONS=true + - LOG_ENABLED=true + - LOG_LEVEL=info + depends_on: + postgres: + condition: service_healthy + postgres-init: + condition: service_completed_successfully + healthcheck: + test: ["CMD", "wget", "-q", "--spider", "http://localhost:3000/_health"] + start_interval: 1s + interval: 10s + timeout: 5s + retries: 5 + start_period: 30s + restart: unless-stopped + networks: [forge-staging] + + # HashiCorp Vault (PERSISTENT, sealed) for hilt's tenant/access-key private + # keys. Integrated Raft storage on the ZFS pool survives restarts; the + # hilt-vault-unseal sidecar unseals it on every (re)start with the key held in + # 1Password. One-time init (operator init + KV v2 mount + storing the unseal + # key/root token in 1Password) is done by `make staging-vault-init`; a + # re-provision wipes /vault/file and requires re-running it. + # See docs/STAGING_DEPLOY.md ("Persistent sealed Vault"). + hilt-vault: + image: ${VAULT_IMAGE} + # Start as root and use the STOCK entrypoint (command: server). When + # docker-entrypoint.sh sees uid 0 it chowns the /vault/file data mount to the + # `vault` user and then su-exec's down to it — so the root-owned host bind dir + # becomes writable with no host-side chown. The `user: "0:0"` is REQUIRED: the + # image's own default user is `vault`, which SKIPS that chown and then can't + # write the bind mount (Raft fails with "permission denied"). The `-dev-*` + # flags the entrypoint injects do NOT enable dev mode (needs `-dev`), so this + # stays a real, sealed server, run by the unprivileged `vault` user. + user: "0:0" + command: server + environment: + - VAULT_ADDR=http://127.0.0.1:8200 + # vault.hcl disables mlock, so no IPC_LOCK capability is needed — and we + # must skip the entrypoint's setcap step: the `setcap` binary isn't present + # in this image and would abort startup ("setcap: not found", exit 127) + # under the entrypoint's `set -e`. + - SKIP_SETCAP=true + volumes: + - ./config/vault/vault.hcl:/vault/config/vault.hcl:ro + - ${FORGE_DATA_DIR}/hilt-vault:/vault/file + # /v1/sys/health returns non-2xx while sealed/uninitialized and 200 only when + # unsealed+active, so this stays unhealthy until the unseal sidecar unseals — + # exactly what hilt's `hilt-vault: service_healthy` dependency should wait for. + healthcheck: + test: ["CMD", "wget", "-q", "--spider", "http://127.0.0.1:8200/v1/sys/health"] + start_interval: 1s + interval: 10s + timeout: 5s + retries: 5 + start_period: 30s + restart: unless-stopped + networks: [forge-staging] + + # Unseal sidecar: a long-running poller that unseals hilt-vault whenever it is + # found sealed — on first boot and after any restart (deploy recreate, box + # reboot, `docker restart hilt-vault`). The unseal key comes from 1Password via + # vault-secrets.env; it is never logged (output is discarded). + hilt-vault-unseal: + image: ${VAULT_IMAGE} + depends_on: + hilt-vault: + # service_started, NOT service_healthy — Vault is unhealthy while sealed, + # and unsealing it is precisely this sidecar's job. + condition: service_started + environment: + - VAULT_ADDR=http://hilt-vault:8200 + - HILT_VAULT_UNSEAL_KEY=${HILT_VAULT_UNSEAL_KEY} + entrypoint: + - /bin/sh + - -c + - | + # `vault status` exit code: 0=unsealed, 2=sealed, 1=error/unreachable. + while true; do + vault status >/dev/null 2>&1; s=$$? + if [ "$$s" = "2" ]; then + vault operator unseal "$$HILT_VAULT_UNSEAL_KEY" >/dev/null 2>&1 || true + fi + sleep 10 + done + restart: unless-stopped + networks: [forge-staging] + + # Tenant management: Fil One Tenant API (partner-key REST) + UCAN RPC (for + # ingot) + did:web document at /.well-known/did.json. Publishes tenant did:plc + # identities to plc and registers tenants as customers with sprue (authority = + # the committed hilt-customer-add proof). + hilt: + image: ${HILT_IMAGE} + user: "0:0" # run as root to read the 0440-perm key file + ports: + - "127.0.0.1:15110:80" + environment: + - HILT_SERVER_HOST=0.0.0.0 + - HILT_SERVER_PORT=80 + - HILT_IDENTITY_KEY_FILE=/keys/hilt.pem + - HILT_IDENTITY_SERVICE_ID=${HILT_DID} + - HILT_AUTH_PARTNER_KEY=${HILT_PARTNER_KEY} + - HILT_STORAGE_TYPE=postgres + - HILT_STORAGE_POSTGRES_DSN=postgres://hilt:${HILT_POSTGRES_PASSWORD}@postgres:5432/hilt?sslmode=disable + - HILT_VAULT_TYPE=hashicorp + - HILT_VAULT_HASHICORP_ADDRESS=http://hilt-vault:8200 + - HILT_VAULT_HASHICORP_AUTH_METHOD=token + - HILT_VAULT_HASHICORP_TOKEN=${HILT_VAULT_TOKEN} + - HILT_PLC_DIRECTORY=http://plc:3000 + - HILT_UPLOAD_SERVICE_ID=${SPRUE_DID} + - HILT_UPLOAD_SERVICE_URL=${SPRUE_PUBLIC_URL} + - HILT_UPLOAD_PRODUCT_ID=${HILT_DID} + - HILT_UPLOAD_PROOFS=/proofs/hilt-customer-add-proof.txt + # NOTE: unlike local dev, no HILT_SERVER_INSECURE_DID_RESOLUTION — staging + # resolves did:web over real https through host Caddy. + - HILT_LOG_LEVEL=info + volumes: + - ${FORGE_SECRETS_DIR}/hilt.pem:/keys/hilt.pem:ro + # Committed delegation proofs (non-secret) come from the git tree. + - ../proofs:/proofs:ro + # NOTE: unlike local dev, there is NO post_start hook registering ingot as + # the regional provider — ingot lives in the piri bundle. Register it + # explicitly after both bundles are healthy: `make staging-register-ingot`. + depends_on: + postgres: + condition: service_healthy + postgres-init: + condition: service_completed_successfully + plc: + condition: service_healthy + hilt-vault: + condition: service_healthy + sprue: + condition: service_healthy + healthcheck: + test: ["CMD", "curl", "-sf", "http://localhost:80/health"] + start_interval: 1s + interval: 10s + timeout: 5s + retries: 5 + start_period: 10s + restart: unless-stopped + networks: [forge-staging] + +# Persistent data lives on the box's ZFS pool (see FORGE_DATA_DIR in config.env), +# not Docker named volumes — so no top-level `volumes:` block. + +networks: + forge-staging: + name: forge-staging-core diff --git a/environments/staging/core/config.env b/environments/staging/core/config.env new file mode 100644 index 0000000..3256811 --- /dev/null +++ b/environments/staging/core/config.env @@ -0,0 +1,26 @@ +# Core bundle — non-secret configuration (committed to git). +# +# Consumed by `docker compose --env-file` for variable interpolation in +# compose.yml and (where referenced) by the rendered config templates. No secret +# values belong here — those live in 1Password and are rendered into secrets.env. + +# Host directory holding rendered configs + key files (provisioned via op, never +# committed). Bind-mounted read-only into the containers. +FORGE_SECRETS_DIR=/root/fil-one/forge/secrets + +# Host directory for persistent container data, on the box's large ZFS pool +# (/mnt/data) rather than the root disk where Docker named volumes live. +FORGE_DATA_DIR=/mnt/data/fil-one/forge + +# Chain id, RPC URL, and all contract addresses live in the shared +# environments/staging/smart-contracts.env (single source of truth) — not here. + +# did:web identities (resolved over HTTPS via host Caddy at *.staging.fil.one). +SPRUE_DID=did:web:sprue.staging.fil.one +SIGNING_SERVICE_DID=did:web:signing-service.staging.fil.one +DELEGATOR_DID=did:web:delegator.staging.fil.one +HILT_DID=did:web:hilt.staging.fil.one + +# Public URLs (Caddy → loopback container ports). +SPRUE_PUBLIC_URL=https://sprue.staging.fil.one +HILT_PUBLIC_URL=https://hilt.staging.fil.one diff --git a/environments/staging/core/config/delegator/delegator.yaml.example b/environments/staging/core/config/delegator/delegator.yaml.example new file mode 100644 index 0000000..5b585f3 --- /dev/null +++ b/environments/staging/core/config/delegator/delegator.yaml.example @@ -0,0 +1,32 @@ +# delegator config — STAGING (REDACTED EXAMPLE, structure only). +# The live file is rendered from delegator.yaml.tpl via `op inject` onto the box. + +server: + host: "0.0.0.0" + port: 80 + +store: + region: "us-east-1" + allowlist_table_name: "delegator-allow-list" + providerinfo_table_name: "delegator-provider-info" + providerweight: 1 + endpoint: "http://dynamodb-local:8000" + +delegator: + key_file: "/keys/delegator.pem" + did: "did:web:delegator.staging.fil.one" + indexing_service_web_did: "did:web:indexer.staging.fil.one" + indexing_service_proof_file: "/proofs/indexing-service-proof.txt" + egress_tracking_service_did: "did:web:etracker.staging.fil.one" + egress_tracking_service_proof_file: "/proofs/egress-tracking-proof.txt" + upload_service_did: "did:web:sprue.staging.fil.one" + +contract: + # All of these are rendered from environments/staging/smart-contracts.env at provision time. + chain_client_endpoint: "" + payments_contract_address: "" + service_contract_address: "" + registry_contract_address: "" + transactor: + chain_id: "" + key: "" diff --git a/environments/staging/core/config/delegator/delegator.yaml.tpl b/environments/staging/core/config/delegator/delegator.yaml.tpl new file mode 100644 index 0000000..607db38 --- /dev/null +++ b/environments/staging/core/config/delegator/delegator.yaml.tpl @@ -0,0 +1,43 @@ +# delegator config — STAGING TEMPLATE. +# +# Rendered with `op inject` into $FORGE_SECRETS_DIR/delegator.yaml and mounted at +# /.delegator.yaml. The only secret is contract.transactor.key (a 1Password reference). The +# delegator identity is a mounted key file; the delegation proofs are committed +# and mounted from the git tree at /proofs. +# +# Rendered with `op inject` (the transactor key) followed by ${VAR} substitution +# from the shared smart-contracts.env (chain id, RPC URL, contract addresses) — so no +# address is duplicated here. +# +# NOTE: the delegator validates an indexing-service and egress-tracking-service +# delegation at startup even though neither service runs in staging — that's why +# those DIDs + proof files are still required here. + +server: + host: "0.0.0.0" + port: 80 + +store: + region: "us-east-1" + allowlist_table_name: "delegator-allow-list" + providerinfo_table_name: "delegator-provider-info" + providerweight: 1 + endpoint: "http://dynamodb-local:8000" + +delegator: + key_file: "/keys/delegator.pem" + did: "did:web:delegator.staging.fil.one" + indexing_service_web_did: "did:web:indexer.staging.fil.one" + indexing_service_proof_file: "/proofs/indexing-service-proof.txt" + egress_tracking_service_did: "did:web:etracker.staging.fil.one" + egress_tracking_service_proof_file: "/proofs/egress-tracking-proof.txt" + upload_service_did: "did:web:sprue.staging.fil.one" + +contract: + chain_client_endpoint: "${LOTUS_RPC_URL}" + payments_contract_address: "${FILECOIN_PAY_ADDRESS}" + service_contract_address: "${FWSS_ADDRESS}" + registry_contract_address: "${SERVICE_PROVIDER_REGISTRY_ADDRESS}" + transactor: + chain_id: ${CHAIN_ID} + key: "{{ op://Fil One/FilOne Forge Staging/delegator-transactor-key }}" diff --git a/environments/staging/core/config/signer/signer.yaml b/environments/staging/core/config/signer/signer.yaml new file mode 100644 index 0000000..217117e --- /dev/null +++ b/environments/staging/core/config/signer/signer.yaml @@ -0,0 +1,11 @@ +# Signing-service config — STAGING. +# +# Contains no secrets and no contract addresses: the signing key and service +# identity key are mounted as separate files, and the command-line flags in +# compose.yml (--rpc-url, --service-contract-address, --service-did, ...) supply +# and override everything chain-related. Those flags read from the shared +# smart-contracts.env, so there is no address duplicated here. Committed and mounted +# read-only from the git tree. + +host: "0.0.0.0" +port: 7446 diff --git a/environments/staging/core/config/sprue/config.yaml.example b/environments/staging/core/config/sprue/config.yaml.example new file mode 100644 index 0000000..b22c201 --- /dev/null +++ b/environments/staging/core/config/sprue/config.yaml.example @@ -0,0 +1,44 @@ +# sprue (upload) config — STAGING (REDACTED EXAMPLE, structure only). +# The live file is rendered from config.yaml.tpl via `op inject` onto the box. + +deployment: + environment: "staging" + allow_provision_without_payment_plan: true + max_replicas: 3 + plc_directory: "http://plc:3000" + +server: + host: "0.0.0.0" + port: 80 + public_url: "https://sprue.staging.fil.one" + +identity: + key_file: "/keys/sprue.pem" + service_did: "did:web:sprue.staging.fil.one" + +indexer: + endpoint: "" + did: "" + +mailer: + type: "nop" + sender: "noreply@staging.fil.one" + +storage: + type: "postgres" + postgres: + dsn: "postgres://sprue:@postgres:5432/sprue?sslmode=disable" + max_conns: 10 + min_conns: 0 + s3: + endpoint: "http://minio:9000" + region: "us-east-1" + use_path_style: true + access_key_id: "" + secret_access_key: "" + agent_message_bucket: "agent-message" + delegation_bucket: "delegation" + upload_shards_bucket: "upload-shards" + +log: + level: "info" diff --git a/environments/staging/core/config/sprue/config.yaml.tpl b/environments/staging/core/config/sprue/config.yaml.tpl new file mode 100644 index 0000000..efd93e2 --- /dev/null +++ b/environments/staging/core/config/sprue/config.yaml.tpl @@ -0,0 +1,56 @@ +# sprue (upload) config — STAGING TEMPLATE. +# +# Rendered with `op inject` into $FORGE_SECRETS_DIR/sprue-config.yaml and mounted +# at /etc/sprue/config.yaml. The Postgres password and the S3/MinIO credentials +# are secrets (1Password references); everything else is literal. sprue reads the +# S3 credentials from storage.s3 below — it does NOT consult the AWS SDK env vars +# for a custom endpoint. + +deployment: + environment: "staging" + # Staging bypasses payment plans (provisioning without a customer payment plan). + allow_provision_without_payment_plan: true + max_replicas: 3 + # did:plc directory for resolving tenant (did:plc) issuers during bucket + # provisioning. Internal to the core bundle — plc has no public route. + plc_directory: "http://plc:3000" + +server: + host: "0.0.0.0" + port: 80 # port 80 for did:web resolution; Caddy fronts TLS + public_url: "https://sprue.staging.fil.one" + +identity: + key_file: "/keys/sprue.pem" + service_did: "did:web:sprue.staging.fil.one" + +# No indexer in staging — an empty endpoint disables it (per sprue config docs). +indexer: + endpoint: "" + did: "" + +# No mailer in staging — "nop" drops outgoing mail (so email-based login is +# unavailable; see the runbook). +mailer: + type: "nop" + sender: "noreply@staging.fil.one" + +storage: + type: "postgres" + postgres: + dsn: "postgres://sprue:{{ op://Fil One/FilOne Forge Staging/sprue-postgres-password }}@postgres:5432/sprue?sslmode=disable" + max_conns: 10 + min_conns: 0 + s3: + endpoint: "http://minio:9000" + region: "us-east-1" + # MinIO requires path-style addressing. + use_path_style: true + access_key_id: "{{ op://Fil One/FilOne Forge Staging/minio-access-key }}" + secret_access_key: "{{ op://Fil One/FilOne Forge Staging/minio-secret-key }}" + agent_message_bucket: "agent-message" + delegation_bucket: "delegation" + upload_shards_bucket: "upload-shards" + +log: + level: "info" diff --git a/environments/staging/core/config/vault/vault.hcl b/environments/staging/core/config/vault/vault.hcl new file mode 100644 index 0000000..0cde6dc --- /dev/null +++ b/environments/staging/core/config/vault/vault.hcl @@ -0,0 +1,32 @@ +# HashiCorp Vault server config for staging hilt-vault — committed, NON-SECRET. +# +# Replaces dev mode: a persistent, properly-sealed Vault using the integrated +# Raft storage backend. Data lives on the box's ZFS pool (bind-mounted at +# /vault/file); the hilt-vault-unseal sidecar unseals it on every (re)start with +# the key held in 1Password. See docs/STAGING_DEPLOY.md ("Persistent sealed +# Vault") and scripts/staging-vault-init.sh (the one-time init ceremony). + +# /vault/file is the image's "blessed" data path: the stock docker-entrypoint.sh +# chowns it to the `vault` user on startup (when the container starts as root) +# before stepping down, so a root-owned host bind mount becomes writable without +# any `user:` override or host-side chown. +storage "raft" { + path = "/vault/file" + node_id = "hilt-vault-0" +} + +listener "tcp" { + address = "0.0.0.0:8200" + tls_disable = 1 +} + +# api_addr/cluster_addr are required by Raft. Container-internal DNS name; Vault +# is not published to the host in staging (network-internal only). +api_addr = "http://hilt-vault:8200" +cluster_addr = "http://hilt-vault:8201" + +# Recommended with integrated storage: Raft manages its own on-disk encryption, +# and disabling mlock avoids the IPC_LOCK/swap-lock requirement. +disable_mlock = true + +ui = false diff --git a/environments/staging/core/secrets.env.example b/environments/staging/core/secrets.env.example new file mode 100644 index 0000000..1451926 --- /dev/null +++ b/environments/staging/core/secrets.env.example @@ -0,0 +1,12 @@ +# Core bundle — secret environment values (REDACTED EXAMPLE, structure only). +# +# The real file is rendered from secrets.env.tpl via `op inject` and lives only on +# the box. This example shows the shape reviewers should expect — no real values. + +POSTGRES_ADMIN_PASSWORD= +SPRUE_POSTGRES_PASSWORD= +HILT_POSTGRES_PASSWORD= +PLC_POSTGRES_PASSWORD= +MINIO_ROOT_USER= +MINIO_ROOT_PASSWORD= +HILT_PARTNER_KEY= diff --git a/environments/staging/core/secrets.env.tpl b/environments/staging/core/secrets.env.tpl new file mode 100644 index 0000000..9788ddc --- /dev/null +++ b/environments/staging/core/secrets.env.tpl @@ -0,0 +1,33 @@ +# Core bundle — SECRET environment values (TEMPLATE). +# +# Rendered on a developer machine with: +# op inject -i secrets.env.tpl (streamed to the box, never written locally) +# producing secrets.env in $FORGE_SECRETS_DIR, consumed via +# docker compose --env-file secrets.env +# +# NEVER commit the rendered secrets.env. Only this template (with 1Password references) is +# tracked. Values resolve from the single 1Password item "FilOne Forge Staging" (vault "Fil One"). +# +# Note: op inject scans the entire file, comments included — so never write a bare +# 1Password reference URL or a template-brace token in a comment; it tries to resolve +# them and fails the whole render. + +# Postgres. One shared instance: the admin superuser initializes the cluster; +# postgres-init creates one role + database per service. SPRUE_POSTGRES_PASSWORD +# must match the password baked into sprue's DSN (rendered sprue-config.yaml). +POSTGRES_ADMIN_PASSWORD={{ op://Fil One/FilOne Forge Staging/core-postgres-admin-password }} +SPRUE_POSTGRES_PASSWORD={{ op://Fil One/FilOne Forge Staging/sprue-postgres-password }} +HILT_POSTGRES_PASSWORD={{ op://Fil One/FilOne Forge Staging/hilt-postgres-password }} +PLC_POSTGRES_PASSWORD={{ op://Fil One/FilOne Forge Staging/plc-postgres-password }} + +# MinIO root credentials. The minio + minio-init containers authenticate with +# these; sprue receives the same key/secret through its rendered config file +# (storage.s3 in sprue-config.yaml), not via these env vars. +MINIO_ROOT_USER={{ op://Fil One/FilOne Forge Staging/minio-access-key }} +MINIO_ROOT_PASSWORD={{ op://Fil One/FilOne Forge Staging/minio-secret-key }} + +# Hilt: pre-shared bearer token for the Tenant API (operators `op read` the same +# field for curl calls). The hilt Vault credentials live in a separate file +# (vault-secrets.env), rendered by `make staging-vault-init` — they are minted at +# runtime by `vault operator init`, not offline like the fields above. +HILT_PARTNER_KEY={{ op://Fil One/FilOne Forge Staging/hilt-partner-key }} diff --git a/environments/staging/core/vault-secrets.env.example b/environments/staging/core/vault-secrets.env.example new file mode 100644 index 0000000..f1bbbfc --- /dev/null +++ b/environments/staging/core/vault-secrets.env.example @@ -0,0 +1,8 @@ +# Core bundle — hilt VAULT secret values (REDACTED EXAMPLE, structure only). +# +# The real file is minted by `vault operator init` and rendered from +# vault-secrets.env.tpl by `make staging-vault-init`; it lives only on the box. +# This example shows the shape reviewers should expect — no real values. + +HILT_VAULT_TOKEN= +HILT_VAULT_UNSEAL_KEY= diff --git a/environments/staging/core/vault-secrets.env.tpl b/environments/staging/core/vault-secrets.env.tpl new file mode 100644 index 0000000..4e06b34 --- /dev/null +++ b/environments/staging/core/vault-secrets.env.tpl @@ -0,0 +1,18 @@ +# Core bundle — hilt VAULT secret values (TEMPLATE). +# +# Separate from secrets.env because these two values have a different lifecycle: +# they are minted at RUNTIME by `vault operator init` (not offline by keygen) and +# stored into 1Password by `make staging-vault-init`, which then renders and ships +# this file. secrets.env is rendered earlier (by staging-provision-core, before +# init has run), so these fields can't live there. +# +# Rendered on a developer machine and streamed to the box, never written locally. +# Consumed via `docker compose --env-file vault-secrets.env` alongside secrets.env. +# +# Note: op inject scans the entire file, comments included — so never write a bare +# 1Password reference URL or a template-brace token in a comment. + +# The Vault root token (from `operator init`) doubles as hilt's client token; the +# hilt-vault-unseal sidecar uses the unseal key to unseal Vault on every restart. +HILT_VAULT_TOKEN={{ op://Fil One/FilOne Forge Staging/hilt-vault-root-token }} +HILT_VAULT_UNSEAL_KEY={{ op://Fil One/FilOne Forge Staging/hilt-vault-unseal-key }} diff --git a/environments/staging/core/versions.env b/environments/staging/core/versions.env new file mode 100644 index 0000000..1948053 --- /dev/null +++ b/environments/staging/core/versions.env @@ -0,0 +1,23 @@ +# Core bundle — pinned image versions (THE atomic deploy unit). +# +# Bump these to roll the staging core stack forward; revert the commit to roll +# back. Image references MUST be immutable: pin every application image to a +# @sha256:... digest before a real deploy. The rolling :main tags below are +# placeholders so `docker compose config` parses — replace them. +# +# Resolve a digest with: +# docker buildx imagetools inspect ghcr.io/fil-forge/sprue:main +# +# Application services (pin to digests): +SPRUE_IMAGE=ghcr.io/fil-forge/sprue:main +SIGNER_IMAGE=ghcr.io/fil-forge/piri-signing-service:main +DELEGATOR_IMAGE=ghcr.io/fil-forge/delegator:main +HILT_IMAGE=ghcr.io/fil-forge/hilt:main +PLC_IMAGE=ghcr.io/fil-forge/did-method-plc:main + +# Dependency containers (a pinned tag is acceptable per the design; digests preferred): +POSTGRES_IMAGE=postgres:16-alpine +MINIO_IMAGE=minio/minio:RELEASE.2024-10-13T13-34-11Z +MC_IMAGE=minio/mc:RELEASE.2024-10-08T09-37-26Z +DYNAMODB_IMAGE=amazon/dynamodb-local:2.5.2 +VAULT_IMAGE=hashicorp/vault:2.0 diff --git a/environments/staging/piri/compose.yml b/environments/staging/piri/compose.yml new file mode 100644 index 0000000..8a8fbe8 --- /dev/null +++ b/environments/staging/piri/compose.yml @@ -0,0 +1,164 @@ +# Forge staging — PIRI bundle. +# +# A single piri storage node (postgres + filesystem) plus ingot (the S3 gateway +# over Forge) and their shared Postgres. Deployed independently of the core +# bundle; talks to core over public https://*.staging.fil.one URLs and to the +# host Lotus node via host-gateway. +# +# Postgres is a single shared instance: a dedicated `admin` superuser plus one +# role + database per service (piri_0, ingot), created by the postgres-init +# one-shot. +# +# Run via the deploy script, which supplies the env files: +# docker compose -p forge-staging-piri \ +# --env-file versions.env --env-file config.env \ +# --env-file $FORGE_SECRETS_DIR/piri-secrets.env up -d +# +# Published ports bind to 127.0.0.1; host Caddy fronts piri-0.staging.fil.one +# and ingot.staging.fil.one. +name: forge-staging-piri + +services: + # Shared Postgres for this bundle. The data dir is piri-postgres (not + # `postgres`) because both bundles share the same FORGE_DATA_DIR root and + # core's instance already owns the `postgres` dir. + postgres: + image: ${POSTGRES_IMAGE} + environment: + - POSTGRES_USER=admin + - POSTGRES_PASSWORD=${POSTGRES_ADMIN_PASSWORD} + volumes: + - ${FORGE_DATA_DIR}/piri-postgres:/var/lib/postgresql/data + healthcheck: + # -h 127.0.0.1 forces a TCP check. On a freshly-wiped data dir the image + # runs a temporary bootstrap server (listen_addresses='') on the unix + # socket only while it runs initdb; a socket-based pg_isready reports + # healthy during that window, so postgres-init connects over TCP too early + # and psql exits 2. A TCP probe stays unready until the real server listens. + test: ["CMD-SHELL", "pg_isready -h 127.0.0.1 -U admin -d admin"] + start_interval: 1s + interval: 5s + timeout: 3s + retries: 10 + start_period: 10s + restart: unless-stopped + networks: [forge-staging-piri] + + # Create the per-service Postgres roles + databases once postgres is up. + # Same structure as the core bundle's postgres-init: idempotent guarded + # CREATEs via \gexec (CREATE DATABASE cannot run inside a DO block) plus an + # unconditional ALTER ROLE so re-provisioned passwords apply. Passwords land + # inside SQL string literals — safe only because keygen secrets are hex. + postgres-init: + image: ${POSTGRES_IMAGE} + depends_on: + postgres: + condition: service_healthy + entrypoint: + - /bin/sh + - -c + - | + set -eu + export PGPASSWORD="$$POSTGRES_ADMIN_PASSWORD" + ensure() { + printf '%s\n' \ + "SELECT 'CREATE ROLE \"$$1\"' WHERE NOT EXISTS (SELECT FROM pg_roles WHERE rolname = '$$1')" \ + '\gexec' \ + "ALTER ROLE \"$$1\" WITH LOGIN PASSWORD '$$2';" \ + "SELECT 'CREATE DATABASE \"$$1\" OWNER \"$$1\"' WHERE NOT EXISTS (SELECT FROM pg_database WHERE datname = '$$1')" \ + '\gexec' \ + | psql -h postgres -U admin -d admin -v ON_ERROR_STOP=1 --quiet + } + ensure piri_0 "$$PIRI_0_POSTGRES_PASSWORD" && + ensure ingot "$$INGOT_POSTGRES_PASSWORD" && + echo "postgres roles + databases ready" + environment: + POSTGRES_ADMIN_PASSWORD: ${POSTGRES_ADMIN_PASSWORD} + PIRI_0_POSTGRES_PASSWORD: ${PIRI_0_POSTGRES_PASSWORD} + INGOT_POSTGRES_PASSWORD: ${INGOT_POSTGRES_PASSWORD} + restart: "no" + networks: [forge-staging-piri] + + piri-0: + image: ${PIRI_IMAGE} + # The piri image runs as USER nobody, but the provisioned key files are + # 0440 root:root and the host data dir is created root:root, so a non-root + # process can neither read the keys nor write /data/piri. Run as root to + # match the local-dev stack (pkg/generate emits user: "0:0" for piri too). + user: "0:0" + ports: + - "127.0.0.1:15100:3000" + extra_hosts: + - "host.docker.internal:host-gateway" + entrypoint: ["/entrypoint.sh"] + environment: + - LOTUS_ENDPOINT=${LOTUS_RPC_URL} + - PUBLIC_URL=${PIRI_PUBLIC_URL} + - PORT=3000 + - HOST=0.0.0.0 + - REGISTRAR_URL=${REGISTRAR_URL} + - OPERATOR_EMAIL=${OPERATOR_EMAIL} + - PIRI_DB_BACKEND=postgres + - PIRI_DB_POSTGRES_URL=postgres://piri_0:${PIRI_0_POSTGRES_PASSWORD}@postgres:5432/piri_0?sslmode=disable + - PIRI_BLOB_BACKEND=filesystem + depends_on: + postgres: + condition: service_healthy + postgres-init: + condition: service_completed_successfully + volumes: + - ${FORGE_SECRETS_DIR}/piri-0.pem:/keys/piri.pem:ro + - ${FORGE_SECRETS_DIR}/piri-0-wallet.hex:/keys/owner-wallet.hex:ro + # Rendered from piri-base-config.toml.tpl at provision time (smart-contracts.env vars). + - ${FORGE_SECRETS_DIR}/piri-base-config.toml:/config/piri-base-config.toml:ro + - ./entrypoint.sh:/entrypoint.sh:ro + - ${FORGE_DATA_DIR}/piri-0:/data/piri + healthcheck: + test: ["CMD-SHELL", "wget -q -O - http://localhost:3000/readyz | grep -qE '\"status\":[[:space:]]*\"ok\"'"] + start_interval: 2s + interval: 10s + timeout: 5s + retries: 30 + start_period: 180s + restart: unless-stopped + networks: [forge-staging-piri] + + # S3 gateway over Forge. Ships object data to sprue (core bundle, public + # https) and authorizes S3 requests against hilt (core bundle, public https); + # its authority toward hilt is the committed hilt-ingot-s3 proof. Reads + # resolve via ingot's local blob_locations registry — no indexer needed. + ingot: + image: ${INGOT_IMAGE} + user: "0:0" # run as root to read the 0440-perm key/config files + command: ["serve", "--config", "/etc/ingot/config.yaml"] + ports: + - "127.0.0.1:15130:9000" + volumes: + - ${FORGE_SECRETS_DIR}/ingot.pem:/keys/ingot.pem:ro + # Rendered from config/ingot/config.yaml.tpl at provision time (op inject). + - ${FORGE_SECRETS_DIR}/ingot-config.yaml:/etc/ingot/config.yaml:ro + # Committed delegation proofs (non-secret) come from the git tree. + - ../proofs:/proofs:ro + # LSM segments, blob spool, token store. + - ${FORGE_DATA_DIR}/ingot:/data + depends_on: + postgres: + condition: service_healthy + postgres-init: + condition: service_completed_successfully + healthcheck: + test: ["CMD", "curl", "-sf", "http://localhost:9000/health"] + start_interval: 1s + interval: 10s + timeout: 5s + retries: 5 + start_period: 10s + restart: unless-stopped + networks: [forge-staging-piri] + +# Persistent data lives on the box's ZFS pool (see FORGE_DATA_DIR in config.env), +# not a Docker named volume. + +networks: + forge-staging-piri: + name: forge-staging-piri diff --git a/environments/staging/piri/config.env b/environments/staging/piri/config.env new file mode 100644 index 0000000..8279a89 --- /dev/null +++ b/environments/staging/piri/config.env @@ -0,0 +1,21 @@ +# Piri bundle — non-secret configuration (committed to git). +# Consumed by `docker compose --env-file` for interpolation in compose.yml. + +# Host directory holding the provisioned piri key files (op-read onto the box). +FORGE_SECRETS_DIR=/root/fil-one/forge/secrets + +# Host directory for persistent piri data (blobs + sqlite), on the box's large +# ZFS pool (/mnt/data) rather than the root disk. +FORGE_DATA_DIR=/mnt/data/fil-one/forge + +# LOTUS_RPC_URL is defined in the shared environments/staging/smart-contracts.env (loaded +# as an env-file at deploy time) so the chain endpoint lives in one place. + +# Core bundle public endpoints (fronted by host Caddy). +REGISTRAR_URL=https://delegator.staging.fil.one + +# This piri node's own public endpoint (Caddy → 127.0.0.1:15100). +PIRI_PUBLIC_URL=https://piri-0.staging.fil.one + +# Operator contact recorded on-chain at registration. +OPERATOR_EMAIL=eng@fil.one diff --git a/environments/staging/piri/config/ingot/config.yaml.example b/environments/staging/piri/config/ingot/config.yaml.example new file mode 100644 index 0000000..77d3f00 --- /dev/null +++ b/environments/staging/piri/config/ingot/config.yaml.example @@ -0,0 +1,30 @@ +# ingot (S3 gateway) config — REDACTED EXAMPLE, structure only. +# +# The real file is rendered from config.yaml.tpl via `op inject` and lives only +# on the box ($FORGE_SECRETS_DIR/ingot-config.yaml). No real values here. + +addr: "0.0.0.0:9000" +region: eu-central-3 +data_dir: /data +log_level: info + +cors_allowed_origins: + - "https://app.fil.one" + - "https://staging.fil.one" + - "https://*.dev.fil.one" + +root_access: "" +root_secret: "" + +identity: + key_file: /keys/ingot.pem + +postgres_dsn: "postgres://ingot:@postgres:5432/ingot?sslmode=disable" + +upload_service_url: "https://sprue.staging.fil.one" +upload_service_did: "did:web:sprue.staging.fil.one" +upload_receipts_url: "https://sprue.staging.fil.one/receipt" + +auth_service_url: "https://hilt.staging.fil.one" +auth_service_did: "did:web:hilt.staging.fil.one" +auth_service_proofs: "/proofs/hilt-ingot-s3-proof.txt" diff --git a/environments/staging/piri/config/ingot/config.yaml.tpl b/environments/staging/piri/config/ingot/config.yaml.tpl new file mode 100644 index 0000000..de2e915 --- /dev/null +++ b/environments/staging/piri/config/ingot/config.yaml.tpl @@ -0,0 +1,53 @@ +# ingot (S3 gateway) config — STAGING TEMPLATE. +# +# Rendered with `op inject` into $FORGE_SECRETS_DIR/ingot-config.yaml and mounted +# at /etc/ingot/config.yaml. The root S3 credentials and the Postgres password +# are secrets (1Password references); everything else is literal. + +# S3 listener. Caddy fronts https://ingot.staging.fil.one → 127.0.0.1:15130. +# PATH-STYLE ONLY: there is no wildcard *.ingot.staging.fil.one DNS/TLS, so +# clients must set force_path_style / addressing_style=path. +addr: "0.0.0.0:9000" +# Must match the region hilt registers ingot under (staging-register-ingot.sh) +# and the AWS_REGION S3 clients sign with. +region: eu-central-3 +data_dir: /data +log_level: info + +# Browser origins the S3 listener answers CORS for (exact origins or subdomain +# wildcards; the '*' replaces the leftmost host label). +cors_allowed_origins: + - "https://app.fil.one" + - "https://staging.fil.one" + - "https://*.dev.fil.one" + +# Root S3 account (break-glass; tenant credentials are minted by hilt). +root_access: "{{ op://Fil One/FilOne Forge Staging/ingot-root-access-key }}" +root_secret: "{{ op://Fil One/FilOne Forge Staging/ingot-root-secret-key }}" + +# Agent identity (the signer that invokes against sprue and hilt). Ingot acts +# under this key's did:key — it serves no did:web document. +identity: + key_file: /keys/ingot.pem + +# Registry / segment metadata — the piri bundle's shared Postgres (role + +# database created by postgres-init). +postgres_dsn: "postgres://ingot:{{ op://Fil One/FilOne Forge Staging/ingot-postgres-password }}@postgres:5432/ingot?sslmode=disable" + +# Forge upload service (sprue) — core bundle, over public https (bundles never +# talk over Docker DNS). +upload_service_url: "https://sprue.staging.fil.one" +upload_service_did: "did:web:sprue.staging.fil.one" +upload_receipts_url: "https://sprue.staging.fil.one/receipt" + +# NOTE: no indexer keys — staging runs no indexing-service, and ingot does not +# need one: reads resolve via its local blob_locations registry (LocalLocator); +# the indexer_endpoint/indexer_did keys don't even exist in ingot's Config. + +# S3 authorization service (hilt, core bundle); the DID is used verbatim (no +# resolution). +auth_service_url: "https://hilt.staging.fil.one" +auth_service_did: "did:web:hilt.staging.fil.one" +# hilt → ingot delegations for /s3/request/authorize and /s3/bucket/* +# (committed environments/staging/proofs/hilt-ingot-s3-proof.txt). +auth_service_proofs: "/proofs/hilt-ingot-s3-proof.txt" diff --git a/environments/staging/piri/config/piri/piri-base-config.toml.example b/environments/staging/piri/config/piri/piri-base-config.toml.example new file mode 100644 index 0000000..f683e7c --- /dev/null +++ b/environments/staging/piri/config/piri/piri-base-config.toml.example @@ -0,0 +1,26 @@ +# Piri base config — STAGING (REDACTED EXAMPLE, structure only). +# The live file is rendered from piri-base-config.toml.tpl by substituting ${VAR} +# from environments/staging/smart-contracts.env (contract addresses, chain id) and +# environments/staging/wallets.env (PAYER_ADDRESS), then shipped to the box. + +[pdp] +chain_id = "" +payer_address = "" + +[pdp.signing_service] +did = "did:web:signing-service.staging.fil.one" +url = "https://signing-service.staging.fil.one" + +[pdp.contracts] +verifier = "" +provider_registry = "" +service = "" +service_view = "" +payments = "" +usdfc_token = "" + +# No [ucan.services.indexer] / [ucan.services.publisher] sections: the +# indexer/IPNI are not deployed in staging and the integration is disabled. +[ucan.services.upload] +did = "did:web:sprue.staging.fil.one" +url = "https://sprue.staging.fil.one" diff --git a/environments/staging/piri/config/piri/piri-base-config.toml.tpl b/environments/staging/piri/config/piri/piri-base-config.toml.tpl new file mode 100644 index 0000000..40d6243 --- /dev/null +++ b/environments/staging/piri/config/piri/piri-base-config.toml.tpl @@ -0,0 +1,33 @@ +# Piri base config — STAGING TEMPLATE (no secrets). +# +# Rendered at provision time by substituting ${VAR} from environments/staging/ +# smart-contracts.env (contract addresses, chain id) and environments/staging/ +# wallets.env (the keygen-written PAYER_ADDRESS), then shipped to the box and +# merged by `piri init`. No address is duplicated — those two files are the +# single source. + +[pdp] +chain_id = "${CHAIN_ID}" +# The funded signing-service payer address (written into wallets.env by +# `smelt staging keygen`). +payer_address = "${PAYER_ADDRESS}" + +[pdp.signing_service] +did = "did:web:signing-service.staging.fil.one" +url = "https://signing-service.staging.fil.one" + +[pdp.contracts] +verifier = "${PDP_VERIFIER_ADDRESS}" +provider_registry = "${SERVICE_PROVIDER_REGISTRY_ADDRESS}" +service = "${FWSS_ADDRESS}" +service_view = "${FWSS_VIEW_ADDRESS}" +payments = "${FILECOIN_PAY_ADDRESS}" +usdfc_token = "${USDFC_TOKEN_ADDRESS}" + +# The indexer/IPNI are NOT deployed in staging, so the integration is disabled: +# omitting [ucan.services.indexer] and ipni_announce_urls makes piri skip claim +# caching and IPNI announcements (blob/accept would otherwise fail hard trying +# to POST to the indexer). Requires piri with optional-indexer support. +[ucan.services.upload] +did = "did:web:sprue.staging.fil.one" +url = "https://sprue.staging.fil.one" diff --git a/environments/staging/piri/entrypoint.sh b/environments/staging/piri/entrypoint.sh new file mode 100755 index 0000000..4740af8 --- /dev/null +++ b/environments/staging/piri/entrypoint.sh @@ -0,0 +1,99 @@ +#!/bin/sh +# Piri entrypoint — STAGING. +# +# Adapted from systems/piri/entrypoint.sh for the split staging deployment. The +# key difference: piri registers via the delegator's registrar HTTP API +# (--registrar-url), NOT by writing directly to DynamoDB. The core bundle's +# DynamoDB is not reachable from the piri bundle, so the dev register-did.sh step +# is intentionally omitted. +set -e + +KEY_FILE="/keys/piri.pem" +WALLET_FILE="/keys/owner-wallet.hex" +BASE_CONFIG="/config/piri-base-config.toml" +DATA_DIR="/data/piri" +TEMP_DIR="/tmp/piri" +CONFIG_FILE="${DATA_DIR}/piri-config.toml" + +LOTUS_ENDPOINT="${LOTUS_ENDPOINT:?LOTUS_ENDPOINT must be set}" +PUBLIC_URL="${PUBLIC_URL:?PUBLIC_URL must be set}" +PORT="${PORT:-3000}" +HOST="${HOST:-0.0.0.0}" +OPERATOR_EMAIL="${OPERATOR_EMAIL:?OPERATOR_EMAIL must be set}" +REGISTRAR_URL="${REGISTRAR_URL:?REGISTRAR_URL must be set}" + +DB_BACKEND="${PIRI_DB_BACKEND:-sqlite}" +BLOB_BACKEND="${PIRI_BLOB_BACKEND:-filesystem}" + +echo "=== Piri Entrypoint (staging) ===" +echo " Lotus: $LOTUS_ENDPOINT" +echo " Public URL: $PUBLIC_URL" +echo " Registrar: $REGISTRAR_URL" +echo " Backends: db=$DB_BACKEND blob=$BLOB_BACKEND" + +mkdir -p "$DATA_DIR" "$TEMP_DIR" + +echo "[1/3] Extracting piri DID..." +# Run the parse on its own line: capturing it inside a `PIRI_DID=$(... | grep)` +# assignment lets `set -e` abort on a grep no-match (or a parse failure) BEFORE +# the checks below run, swallowing the real error — e.g. an unreadable key file +# reports nothing but the silent crash-loop. Separating the steps surfaces the +# actual stderr (`|| true` on grep so we reach the explicit, informative checks). +if ! PARSE_OUTPUT=$(/usr/bin/piri identity parse "$KEY_FILE" 2>&1); then + echo "ERROR: 'piri identity parse $KEY_FILE' failed:" >&2 + echo "$PARSE_OUTPUT" >&2 + exit 1 +fi +PIRI_DID=$(printf '%s\n' "$PARSE_OUTPUT" | grep -oE 'did:key:z[a-zA-Z0-9]+' || true) +if [ -z "$PIRI_DID" ]; then + echo "ERROR: no did:key found in 'piri identity parse $KEY_FILE' output:" >&2 + echo "$PARSE_OUTPUT" >&2 + exit 1 +fi +echo " DID: $PIRI_DID" + +echo "[2/3] Initializing piri..." +if [ -f "$CONFIG_FILE" ] && grep -q "proof_set" "$CONFIG_FILE" 2>/dev/null; then + echo " Config exists, skipping init" +else + [ -f "$CONFIG_FILE" ] && rm -f "$CONFIG_FILE" + cd "$DATA_DIR" + # Build the init command as a positional-parameter list rather than a string + # run through `eval`: values reach piri as literal argv entries, so spaces or + # shell metacharacters in any of them can't word-split or inject. + set -- /usr/bin/piri init \ + --base-config="$BASE_CONFIG" \ + --registrar-url="$REGISTRAR_URL" \ + --data-dir="$DATA_DIR" \ + --temp-dir="$TEMP_DIR" \ + --key-file="$KEY_FILE" \ + --wallet-file="$WALLET_FILE" \ + --lotus-endpoint="$LOTUS_ENDPOINT" \ + --public-url="$PUBLIC_URL" \ + --port="$PORT" \ + --host="$HOST" \ + --operator-email="$OPERATOR_EMAIL" + + if [ "$DB_BACKEND" = "postgres" ]; then + set -- "$@" \ + --db-type=postgres \ + --db-postgres-url="${PIRI_DB_POSTGRES_URL:?PIRI_DB_POSTGRES_URL required for postgres backend}" + fi + if [ "$BLOB_BACKEND" = "s3" ]; then + set -- "$@" \ + --s3-endpoint="${PIRI_S3_ENDPOINT:?PIRI_S3_ENDPOINT required for s3 backend}" \ + --s3-bucket-prefix="${PIRI_S3_BUCKET_PREFIX:-piri-0-}" \ + --s3-access-key-id="${PIRI_S3_ACCESS_KEY_ID:?PIRI_S3_ACCESS_KEY_ID required}" \ + --s3-secret-access-key="${PIRI_S3_SECRET_ACCESS_KEY:?PIRI_S3_SECRET_ACCESS_KEY required}" + fi + + "$@" + echo " Init complete" +fi + +echo "[3/3] Starting piri..." +# No extra args: the config file init wrote carries everything (backends, DSN, +# proof set). NOTE: `$@` must NOT be passed here — the init branch above rebuilt +# it via `set --`, so on a first boot it still holds the whole `piri init` argv +# and `serve full` dies on `unknown flag: --base-config`. +exec /usr/bin/piri serve full --config "$CONFIG_FILE" diff --git a/environments/staging/piri/secrets.env.example b/environments/staging/piri/secrets.env.example new file mode 100644 index 0000000..489fdbf --- /dev/null +++ b/environments/staging/piri/secrets.env.example @@ -0,0 +1,9 @@ +# Piri bundle — secret environment values (REDACTED EXAMPLE, structure only). +# +# The real file is rendered from secrets.env.tpl via `op inject` and lives only on +# the box as $FORGE_SECRETS_DIR/piri-secrets.env. This example shows the shape +# reviewers should expect — no real values. + +POSTGRES_ADMIN_PASSWORD= +PIRI_0_POSTGRES_PASSWORD= +INGOT_POSTGRES_PASSWORD= diff --git a/environments/staging/piri/secrets.env.tpl b/environments/staging/piri/secrets.env.tpl new file mode 100644 index 0000000..ba3a01a --- /dev/null +++ b/environments/staging/piri/secrets.env.tpl @@ -0,0 +1,23 @@ +# Piri bundle — SECRET environment values (TEMPLATE). +# +# Rendered on a developer machine with: +# op inject -i secrets.env.tpl (streamed to the box, never written locally) +# producing piri-secrets.env in $FORGE_SECRETS_DIR (a distinct basename — the +# core bundle owns secrets.env), consumed via +# docker compose --env-file $FORGE_SECRETS_DIR/piri-secrets.env +# +# NEVER commit the rendered piri-secrets.env. Only this template (with 1Password +# references) is tracked. Values resolve from the single 1Password item +# "FilOne Forge Staging" (vault "Fil One"). +# +# Note: op inject scans the entire file, comments included — so never write a bare +# 1Password reference URL or a template-brace token in a comment; it tries to resolve +# them and fails the whole render. + +# Postgres. One shared instance: the admin superuser initializes the cluster; +# postgres-init creates one role + database per service (piri_0, ingot). +# INGOT_POSTGRES_PASSWORD must match the password baked into ingot's DSN +# (rendered ingot-config.yaml). +POSTGRES_ADMIN_PASSWORD={{ op://Fil One/FilOne Forge Staging/piri-postgres-admin-password }} +PIRI_0_POSTGRES_PASSWORD={{ op://Fil One/FilOne Forge Staging/piri-0-postgres-password }} +INGOT_POSTGRES_PASSWORD={{ op://Fil One/FilOne Forge Staging/ingot-postgres-password }} diff --git a/environments/staging/piri/versions.env b/environments/staging/piri/versions.env new file mode 100644 index 0000000..331ef65 --- /dev/null +++ b/environments/staging/piri/versions.env @@ -0,0 +1,12 @@ +# Piri bundle — pinned image versions (THE atomic deploy unit for this bundle). +# +# Pin application images to an immutable @sha256:... digest before a real +# deploy; the rolling :main tags below are placeholders so `docker compose +# config` parses. +# docker buildx imagetools inspect ghcr.io/fil-forge/piri:main +# +PIRI_IMAGE=ghcr.io/fil-forge/piri:main +INGOT_IMAGE=ghcr.io/fil-forge/ingot:main + +# Dependency containers (a pinned tag is acceptable per the design; digests preferred): +POSTGRES_IMAGE=postgres:16-alpine diff --git a/environments/staging/proofs/.gitkeep b/environments/staging/proofs/.gitkeep new file mode 100644 index 0000000..22fd532 --- /dev/null +++ b/environments/staging/proofs/.gitkeep @@ -0,0 +1,3 @@ +# UCAN delegation proofs are written here by `smelt staging keygen` and committed: +# piri-0-proof.txt indexing-service-proof.txt egress-tracking-proof.txt +# hilt-customer-add-proof.txt hilt-ingot-s3-proof.txt diff --git a/environments/staging/proofs/egress-tracking-proof.txt b/environments/staging/proofs/egress-tracking-proof.txt new file mode 100644 index 0000000..d4508c8 Binary files /dev/null and b/environments/staging/proofs/egress-tracking-proof.txt differ diff --git a/environments/staging/proofs/hilt-customer-add-proof.txt b/environments/staging/proofs/hilt-customer-add-proof.txt new file mode 100644 index 0000000..f955ef5 --- /dev/null +++ b/environments/staging/proofs/hilt-customer-add-proof.txt @@ -0,0 +1 @@ +EH4sIAAAAAAAA/1qYllySp1tm2BjxvynC4f7nTfseZde6SIX8Ul20zvqyX9CkBt/un8uF3s3Z+SGc+fCVg62awYszevTOMa54qK+/T/jF0m93CtiP/giZPLH8kzHXosQMDxPGt4xvGYULi0uTE/P0U3LSHQz1DPQMdIuS9QyXJyeWplTIpGSmWJWnJlllZOaU6BWXJKZn5qXrpWXm6OXnpSYn56bk6ieXFpfk56YW6SempCSnVhR8S84sLq6QhWksLigqTcXQWZCf05BcXJpEQF1qXn5ecmrArjl/Hwh3h+816ZrO9Gb11fuAAAAA///a4DQ/CgEAAA== diff --git a/environments/staging/proofs/hilt-ingot-s3-proof.txt b/environments/staging/proofs/hilt-ingot-s3-proof.txt new file mode 100644 index 0000000..8f39ae0 --- /dev/null +++ b/environments/staging/proofs/hilt-ingot-s3-proof.txt @@ -0,0 +1 @@ +EH4sIAAAAAAAA/1qYllySp1tm2BrJKNsU4aBZXVVTsGr+5tQS/60M209pqUx3vFB46XW5qopky4F1f1aH/am3Ei79d2Gv+c14lecbBdo3me1e9KKdb8vjOxNKN2/jXZSY4WHC+JbxLaNwYXFpcmKefkpOuoOhnoGegW5Rsp7h8uTE0pQKi5TMFKvs1EqrKjPf7FLjvAx3Y8uK/OCgAOMy8wx33yLjYtOoUp+gkOSo3JCsNGeTkpDyDNfEEKfk5OTclEL9YmP9pNLk7NQS/eSi1MSS1OTUioJvyZnFxRUyIIPLU5OsMjJzSvSKSxLTM/PS9dIyc/Ty81KTC/JzGpKLS5PwK0vNy89LTg34eOiTU1xIEL9tqFK86+R9jyIZFZsiHOJs79+cdNP6WmfQJp9upv9BAkfMJex9Dl3muyjo+a9yspzJMxfPWn+xnM5vJyPEwwXmRny5xXswXmdSxdqi6HVyq9joED6loPApSi0sTS0u0U8sLcnIL8qsok0YBW5TPBP+60O+xp1HVzt3ml+KZJRuinBI3Kkpu+pjyepnbtbWvQefWb76cPdq9I0DxvlRDOfm7btRmrtaeSP74Zsnpngm8ARI7dm9barBfm+WlR+EKysE+1Jc6ZGG8pHSUGZeWj5NQkdycXvn11DtzIOfXhdueMzeBclhqV2yvVdF5+tOWh7snBTp581t/V3sxtXbelw9E5d2NRj+/qRc4Cv++3bZp+jYvlnvpvyzklrat2hz7ema3CV1csdlOeicw1JSc1JplMMkpq/b+PBKx7bc1jxT/8nSsyCppzcia/ER4dU/fznlpYamdzmk2l6ZdjbETV+46YTJ665ngf+PyByPYmR0aTz43edX92733d+7+xOFX/FuW/zB9w+bFwOdU09OZnEJTULH+/WCqtcrnzbrujFOL5q7MxsQAAD//1gegIapBQAA diff --git a/environments/staging/proofs/indexing-service-proof.txt b/environments/staging/proofs/indexing-service-proof.txt new file mode 100644 index 0000000..dd2a441 Binary files /dev/null and b/environments/staging/proofs/indexing-service-proof.txt differ diff --git a/environments/staging/proofs/piri-0-proof.txt b/environments/staging/proofs/piri-0-proof.txt new file mode 100644 index 0000000..b1aa3c6 --- /dev/null +++ b/environments/staging/proofs/piri-0-proof.txt @@ -0,0 +1 @@ +EH4sIAAAAAAAA/1qYllySp1tm2BLJaNoU4SC0pNdH8N/L/TmS0XGyrj7TJ87flywYnvyCaVem4POHborWJYb9Cf5nud6WBYgdVfRNN89rvHF4ToSgsdTD5iMWT/gXJWZ4mDC+ZXzLKFxYXJqcmKefkpPuYKhnoGegW5SsZ7g8ObE0pUI2JTPFqjw1yaq4oKg0Va+4JDE9My9dLy0zRy8/LzU5OTclRz8pJz9JPzE5ObWgJDm1ouBbcmZxcYUFSGN2aqVVlZlvdkZQimt2hYt7RLare1h5sllJWISzkWGER0B5aFVJiUtWRFJwcUFEULaPj7NLckF+TkNycWkS+Uak5uXnJacG+J1OmvKgcYbw9niV/4Y8E+0iGe2bIhySfiTJ7Jrw8ktxtehh0f5H85QdM+dbad77NSP5XLGyiG3f3JAVMze2TlH8sGi3WI348/XP69wS48VT7H4v/nux8fUVJmqFWxkk3IpSC3IykxP1E3Ny8pMTS1IHVQhWsKWdUPeZ4+ul38jT29LyJ5LRvCnCIWv1DzVV9r+65iekXjVlbtZ67fb1RtJyxZsiu+bv765rM1Hmcy87o7X+7fUZX7ZddQll/D2nq7enRfDl1XWtR9UdF7JSKwTzoClvMIacI7PaFwvps7Pq17AnLRVZ8iyS0agpwuHep56pglXuuWv/3MgUqNox7eTR8moRI8kM3qUr1a/qm/KL9Ty/56HV5Xh47v7zz2cKMlTnunx6ysC1IVZnRyJ7fBovtUIuU78gpUA/My8tf1AF2o6JFw33+0RcvLmkd/7yPaf3AwIAAP//R3JfUfIEAAA= diff --git a/environments/staging/smart-contracts.env b/environments/staging/smart-contracts.env new file mode 100644 index 0000000..acb03e3 --- /dev/null +++ b/environments/staging/smart-contracts.env @@ -0,0 +1,35 @@ +# Forge staging — smart-contract & chain configuration (committed). SINGLE SOURCE OF TRUTH. +# +# Shared by BOTH bundles. Every contract address, the chain id, and the RPC URL +# live here exactly once; config files reference them as ${VAR} and the provision +# step renders them in. Loaded as a `--env-file` at deploy time for the variables +# that compose interpolates directly (e.g. the signing-service --rpc-url and +# --service-contract-address flags). +# +# Calibration testnet (314159). Addresses are the deployed *proxy* contracts. +# +# The FilOz Contracts: https://github.com/FilOzone/filecoin-services/releases +# The Storacha Contracts: https://github.com/fil-forge/filecoin-services +# +# The list of contract address that the Forge network last used can be found here: +# https://github.com/fil-forge/filecoin-services/blob/main/service_contracts/deployments.json#L19-L36 + +CHAIN_ID=314159 + +# Host Lotus Eth RPC via host-gateway, over WebSocket. Must be ws://, not http://: +# piri watches for tx confirmations via ChainNotify, a streaming subscription that +# only delivers events over a WebSocket — over plain HTTP the scheduler never fires +# and receipts sit pending forever. We use ws:// (not wss://) because the localhost +# node is plaintext with no TLS in front of it. +LOTUS_RPC_URL=ws://host.docker.internal:1234/rpc/v1 + +# Forge contract addresses (Calibnet proxies). +PDP_VERIFIER_ADDRESS=0x85e366Cf9DD2c0aE37E963d9556F5f4718d6417C +FWSS_ADDRESS=0x0c6875983B20901a7C3c86871f43FdEE77946424 +FWSS_VIEW_ADDRESS=0xEAD67d775f36D1d2894854D20e042C77A3CC20a5 +SERVICE_PROVIDER_REGISTRY_ADDRESS=0x839e5c9988e4e9977d40708d0094103c0839Ac9D +FILECOIN_PAY_ADDRESS=0x09a0fDc2723fAd1A7b8e3e00eE5DF73841df55a0 +USDFC_TOKEN_ADDRESS=0xb3042734b608a1B16e9e86B374A3f3e389B4cDf0 + +# Wallet addresses (incl. PAYER_ADDRESS, which renders into piri's base config) +# live in the sibling wallets.env, written by `smelt staging keygen`. diff --git a/environments/staging/wallets.env b/environments/staging/wallets.env new file mode 100644 index 0000000..b758a64 --- /dev/null +++ b/environments/staging/wallets.env @@ -0,0 +1,12 @@ +# Forge staging — wallet addresses (committed, public). NOT secrets: the private +# keys live in the 1Password item op://Fil One/FilOne Forge Staging. These public +# addresses are kept here so they're easy to find when topping up balances (a +# periodic chore on Calibnet). +# +# WRITTEN BY `smelt staging keygen` — do not edit by hand. Each must be funded via +# the Calibnet faucet. PAYER_ADDRESS additionally renders into piri's base config +# at provision time. + +PAYER_ADDRESS=0x1B900964cf6D0B4622376C48611A0eACa6Af61ED +DELEGATOR_TRANSACTOR_ADDRESS=0xdA857029A0FeEefa528b5f96a2942329b5C5f60A +PIRI_0_OWNER_ADDRESS=0xDf4e39AFF9BcAcb1fAf6B2A16663DC35Ec2a05Ad diff --git a/go.mod b/go.mod index df6b3e7..3a99320 100644 --- a/go.mod +++ b/go.mod @@ -11,6 +11,8 @@ require ( github.com/spf13/cobra v1.10.2 github.com/testcontainers/testcontainers-go v0.42.0 github.com/testcontainers/testcontainers-go/modules/compose v0.42.0 + gitlab.com/yawning/secp256k1-voi v0.0.0-20230925100816-f2616030848b + golang.org/x/crypto v0.50.0 golang.org/x/sync v0.20.0 gopkg.in/yaml.v3 v3.0.1 ) @@ -137,6 +139,7 @@ require ( github.com/whyrusleeping/cbor-gen v0.3.1 // indirect github.com/xhit/go-str2duration/v2 v2.1.0 // indirect github.com/yusufpapurcu/wmi v1.2.4 // indirect + gitlab.com/yawning/tuplehash v0.0.0-20230713102510-df83abbf9a02 // indirect go.opentelemetry.io/auto/sdk v1.2.1 // indirect go.opentelemetry.io/contrib/instrumentation/google.golang.org/grpc/otelgrpc v0.63.0 // indirect go.opentelemetry.io/contrib/instrumentation/net/http/httptrace/otelhttptrace v0.63.0 // indirect @@ -154,7 +157,6 @@ require ( go.opentelemetry.io/proto/otlp v1.10.0 // indirect go.yaml.in/yaml/v3 v3.0.4 // indirect go.yaml.in/yaml/v4 v4.0.0-rc.4 // indirect - golang.org/x/crypto v0.50.0 // indirect golang.org/x/exp v0.0.0-20260312153236-7ab1446f8b90 // indirect golang.org/x/net v0.53.0 // indirect golang.org/x/sys v0.43.0 // indirect diff --git a/pkg/generate/keys.go b/pkg/generate/keys.go index e7bab4a..846659f 100644 --- a/pkg/generate/keys.go +++ b/pkg/generate/keys.go @@ -67,6 +67,13 @@ func GenerateKeys(keysDir string, nodes []manifest.ResolvedPiriNode, force bool) return nil } +// GenerateEd25519Key generates an Ed25519 service identity key pair (name.pem + +// name.pub) in the same PKCS8 PEM format used by the local dev stack. Exported so +// the staging keygen (pkg/staging) reuses the identical format the services expect. +func GenerateEd25519Key(keysDir, name string, force bool) error { + return generateEd25519Key(keysDir, name, force) +} + // generateEd25519Key generates an Ed25519 key pair in PEM format, plus a // .did file holding the key's did:key identifier (consumed by shell // scripts that have no DID tooling, e.g. systems/hilt/post_start.sh). diff --git a/pkg/staging/keygen.go b/pkg/staging/keygen.go new file mode 100644 index 0000000..755f799 --- /dev/null +++ b/pkg/staging/keygen.go @@ -0,0 +1,355 @@ +// Package staging implements the secret-generation ceremony for the Forge +// staging deployment. Unlike the local dev stack (pkg/generate), which +// regenerates throwaway keys on every `make up` and derives EVM wallets from +// Anvil's public deterministic accounts, staging secrets are long-lived: they +// are stored in 1Password and never rotated per-deploy. +// +// The ceremony has "ensure" semantics and is safe to re-run: every secret the +// 1Password item already holds is reused byte-for-byte (funded wallets and +// registered DIDs survive), and only missing fields are generated and added. +// To rotate a specific secret, delete its field from the 1Password item (and +// any proof files signed with it) and re-run. +// +// What this produces: +// - Ed25519 service identity keys (PEM) for every service identity. +// - Real, random secp256k1 EVM wallets for the chain-transacting roles +// (signing-service payer, delegator transactor, piri owner). Their addresses +// are printed so the operator can fund them from a Calibnet faucet. +// - Random connection secrets (Postgres passwords, S3 access/secret keys, +// hilt partner key, vault token, ingot root S3 credentials). +// - UCAN delegation proofs (committed to git) signed with the identity keys, +// using the staging did:web identities. A proof is re-issued only when +// missing or when a key it depends on was freshly generated. +// +// Private key material is written only to a temp dir, copied into 1Password, +// then wiped. Only public EVM addresses are ever printed. Nothing secret is +// logged or committed. +package staging + +import ( + "crypto/rand" + "encoding/hex" + "fmt" + "os" + "path/filepath" + "strings" + + "github.com/fil-forge/libforge/identity" + "github.com/fil-forge/smelt/pkg/generate" + "github.com/fil-forge/ucantone/multikey" +) + +// Staging did:web identities. The upload service is published as "sprue". +const ( + DIDSprue = "did:web:sprue.staging.fil.one" + DIDSigningService = "did:web:signing-service.staging.fil.one" + DIDDelegator = "did:web:delegator.staging.fil.one" + DIDIndexer = "did:web:indexer.staging.fil.one" // not deployed; identity used only to sign the proof the delegator requires + DIDEtracker = "did:web:etracker.staging.fil.one" // not deployed; identity used only to sign the proof the delegator requires + DIDHilt = "did:web:hilt.staging.fil.one" + // ingot has no did:web — it acts under its did:key (derived from ingot.pem). +) + +// serviceIdentityKeys are the Ed25519 identities generated for staging. indexer +// and etracker are NOT run as containers, but the delegator validates an +// indexing- and egress-service delegation at startup, so we still need their +// keys to issue those proofs. +var serviceIdentityKeys = []string{ + "sprue", + "signing-service", + "delegator", + "indexer", + "etracker", + "piri-0", + "guppy", + "hilt", + "ingot", +} + +// connSecretFields are the random connection secrets for the dependency +// containers and service-to-service auth. Each becomes one CONCEALED 1Password +// field holding a randomHex value (hex-only by construction — several land +// inside JSON env values and Postgres DSNs, where quoting is not an option). +var connSecretFields = []string{ + // core bundle + "core-postgres-admin-password", + "sprue-postgres-password", + "hilt-postgres-password", + "plc-postgres-password", + "hilt-partner-key", + // NOTE: the hilt Vault credentials (unseal key + root token) are NOT minted + // here. Vault's Shamir unseal key and root token are produced by `vault + // operator init` at runtime and cannot be generated offline, so they are + // minted and stored in 1Password by `make staging-vault-init` instead. + "minio-access-key", + "minio-secret-key", + // piri bundle + "piri-postgres-admin-password", + "piri-0-postgres-password", + "ingot-postgres-password", + "ingot-root-access-key", + "ingot-root-secret-key", +} + +// Options controls the keygen ceremony. +type Options struct { + ProjectDir string // repo root; proofs are written under environments/staging/proofs + OPVault string // 1Password vault (e.g. "Fil One") + OPItem string // 1Password item title (e.g. "FilOne Forge Staging") + Store bool // reuse/store secrets in 1Password via the op CLI + Proofs bool // generate UCAN delegation proofs (requires ucantool) + Ucantool string // ucantool binary name/path +} + +// Result reports the non-secret outcome of the ceremony. +type Result struct { + // FundAddresses maps a role to the 0x EVM address that must be funded. + FundAddresses map[string]string + ProofsWritten []string + OPFields []string // field names written to 1Password (names only, never values) + // ReusedFields / GeneratedFields split the field names by whether the value + // was read back from the existing 1Password item or freshly generated this + // run. An all-reused run means nothing was rotated. + ReusedFields []string + GeneratedFields []string + // WalletsEnvPath is the wallets.env file the wallet addresses were written + // into (empty if not written). + WalletsEnvPath string +} + +// Keygen ensures the staging keys, secrets, and proofs exist. It is idempotent +// against 1Password: fields the item already holds are reused unchanged (so +// funded wallets, registered DIDs, and shipped keys survive a re-run), and only +// missing fields are generated and added. With Store disabled it always +// generates fresh values and stores nothing — useful only for dry runs. +func Keygen(opts Options) (*Result, error) { + if opts.Ucantool == "" { + opts.Ucantool = "ucantool" + } + proofsDir := filepath.Join(opts.ProjectDir, "environments", "staging", "proofs") + if err := os.MkdirAll(proofsDir, 0o755); err != nil { + return nil, fmt.Errorf("create proofs dir: %w", err) + } + + // All private material lands in a temp dir we wipe at the end. It is never + // written under the repo, so nothing secret can be accidentally committed. + tmp, err := os.MkdirTemp("", "smelt-staging-keygen-") + if err != nil { + return nil, fmt.Errorf("create temp dir: %w", err) + } + defer os.RemoveAll(tmp) + + res := &Result{FundAddresses: map[string]string{}} + + // 0. Read back whatever the 1Password item already holds. This must happen + // before any generation: minting fresh keys while 1Password holds + // different ones would silently rotate the deployed identity set. The + // read fails loudly if the op CLI is missing or unauthenticated. + existing := map[string]string{} + if opts.Store { + existing, err = readOnePasswordFields(opts.OPVault, opts.OPItem) + if err != nil { + return nil, fmt.Errorf("read existing 1Password item: %w", err) + } + } + fresh := map[string]bool{} + // reuse writes an existing field value to its temp file; it returns false + // when the field is not in the item yet (caller generates it instead). + reuse := func(field, file string, mode os.FileMode) (bool, error) { + value, ok := existing[field] + if !ok { + fresh[field] = true + res.GeneratedFields = append(res.GeneratedFields, field) + return false, nil + } + if err := os.WriteFile(filepath.Join(tmp, file), []byte(value), mode); err != nil { + return false, fmt.Errorf("write reused %s: %w", field, err) + } + res.ReusedFields = append(res.ReusedFields, field) + return true, nil + } + + // 1. Ed25519 service identities (reuses the exact PKCS8 PEM format the + // services expect — see pkg/generate.GenerateEd25519Key). + for _, name := range serviceIdentityKeys { + reused, err := reuse(name+"-key", name+".pem", 0o600) + if err != nil { + return nil, err + } + if !reused { + if err := generate.GenerateEd25519Key(tmp, name, true); err != nil { + return nil, fmt.Errorf("generate %s identity: %w", name, err) + } + } + } + + // 2. Real random EVM wallets for the chain-transacting roles. Each role gets + // its own wallet so the services never collide on transaction nonces. + walletFiles := map[string]string{} // op field name -> temp file path + wallets := []struct { + role string // human label for funding + opField string // 1Password field name + file string // temp file basename + addrVar string // wallets.env variable holding the public address + contents func(*EVMWallet) string + parse func(string) (*EVMWallet, error) + }{ + {"signing-service payer", "payer-key", "payer-key.hex", "PAYER_ADDRESS", (*EVMWallet).RawHex, ParseEVMWalletRawHex}, + {"delegator transactor", "delegator-transactor-key", "delegator-transactor-key.hex", "DELEGATOR_TRANSACTOR_ADDRESS", (*EVMWallet).Hex0x, ParseEVMWalletHex0x}, + {"piri-0 owner", "piri-0-wallet", "piri-0-wallet.hex", "PIRI_0_OWNER_ADDRESS", (*EVMWallet).PiriWalletHex, ParseEVMWalletPiriHex}, + } + // Wallet addresses are public; record them in the committed wallets.env so they + // are easy to find for periodic balance top-ups. (Private keys go to 1Password.) + walletsEnv := filepath.Join(opts.ProjectDir, "environments", "staging", "wallets.env") + for _, w := range wallets { + var wallet *EVMWallet + if value, ok := existing[w.opField]; ok { + // Reuse the funded wallet; re-derive its public address so a stale + // or missing wallets.env entry heals itself. + wallet, err = w.parse(value) + if err != nil { + return nil, fmt.Errorf("parse existing %s wallet from 1Password: %w", w.role, err) + } + res.ReusedFields = append(res.ReusedFields, w.opField) + } else { + wallet, err = GenerateEVMWallet() + if err != nil { + return nil, fmt.Errorf("generate %s wallet: %w", w.role, err) + } + fresh[w.opField] = true + res.GeneratedFields = append(res.GeneratedFields, w.opField) + } + path := filepath.Join(tmp, w.file) + if err := os.WriteFile(path, []byte(w.contents(wallet)), 0o600); err != nil { + return nil, fmt.Errorf("write %s wallet: %w", w.role, err) + } + walletFiles[w.opField] = path + res.FundAddresses[w.role] = wallet.Address + if err := upsertEnvVar(walletsEnv, w.addrVar, wallet.Address); err != nil { + return nil, fmt.Errorf("write %s to %s: %w", w.addrVar, walletsEnv, err) + } + } + res.WalletsEnvPath = walletsEnv + + // 3. Connection secrets for the dependency containers we run (Postgres, + // MinIO, Vault) and service-to-service auth (hilt partner key, ingot + // root S3 credentials). + connFiles := map[string]string{} + for _, field := range connSecretFields { + reused, err := reuse(field, field, 0o600) + if err != nil { + return nil, err + } + if !reused { + secret, err := randomHex(24) + if err != nil { + return nil, fmt.Errorf("generate %s: %w", field, err) + } + if err := os.WriteFile(filepath.Join(tmp, field), []byte(secret), 0o600); err != nil { + return nil, fmt.Errorf("write %s: %w", field, err) + } + } + connFiles[field] = filepath.Join(tmp, field) + } + + // 4. UCAN delegation proofs (committed to git), signed with the staging + // identities and addressed to the staging did:web audiences (plus ingot's + // did:key). Proofs are re-issued only when missing or when a key they + // depend on was freshly generated above. + if opts.Proofs { + ingotDID, err := deriveDIDFromPEM(filepath.Join(tmp, "ingot.pem")) + if err != nil { + return nil, fmt.Errorf("derive ingot did:key: %w", err) + } + written, err := generateProofs(opts.Ucantool, tmp, proofsDir, ingotDID, fresh) + if err != nil { + return nil, fmt.Errorf("generate proofs: %w", err) + } + res.ProofsWritten = written + } + + // 5. Store everything secret in the single 1Password item. Identity PEMs, + // wallets, and connection secrets are read from their temp files and written + // into a JSON template (see storeInOnePassword) so multi-line PEMs and + // special characters survive intact and no value lands on a command line. + // Reused values are re-written byte-for-byte (a no-op edit); fresh values + // are added — the item is never partially refreshed. + if opts.Store { + fields := map[string]string{} + for _, name := range serviceIdentityKeys { + fields[name+"-key"] = filepath.Join(tmp, name+".pem") + } + for field, path := range walletFiles { + fields[field] = path + } + for field, path := range connFiles { + fields[field] = path + } + names, err := storeInOnePassword(opts.OPVault, opts.OPItem, fields) + if err != nil { + return nil, fmt.Errorf("store secrets in 1Password: %w", err) + } + res.OPFields = names + } + + return res, nil +} + +// deriveDIDFromPEM computes the did:key identifier for an Ed25519 private key +// PEM — the same derivation pkg/generate uses for the local .did files. +func deriveDIDFromPEM(pemPath string) (string, error) { + pemBytes, err := os.ReadFile(pemPath) + if err != nil { + return "", err + } + signer, err := identity.DecodeSignerFromPEM(pemBytes) + if err != nil { + return "", fmt.Errorf("decode private key: %w", err) + } + return multikey.KeyIssuer(signer).DID().String(), nil +} + +// upsertEnvVar sets KEY=value in a dotenv-style file: it replaces an existing +// `KEY=...` line in place (preserving everything else) or appends one. Creates +// the file if it does not exist. The output always ends with a single trailing +// newline, regardless of whether the input had one. +func upsertEnvVar(path, key, value string) error { + prefix := key + "=" + line := prefix + value + + data, err := os.ReadFile(path) + if err != nil { + if os.IsNotExist(err) { + return os.WriteFile(path, []byte(line+"\n"), 0o644) + } + return err + } + + // Drop the trailing newline before splitting so a final "\n" doesn't yield an + // empty trailing element; we re-add exactly one newline when writing back. + var lines []string + if content := strings.TrimSuffix(string(data), "\n"); content != "" { + lines = strings.Split(content, "\n") + } + found := false + for i, l := range lines { + if strings.HasPrefix(l, prefix) { + lines[i] = line + found = true + break + } + } + if !found { + lines = append(lines, line) + } + return os.WriteFile(path, []byte(strings.Join(lines, "\n")+"\n"), 0o644) +} + +// randomHex returns n random bytes hex-encoded (2n chars). +func randomHex(n int) (string, error) { + b := make([]byte, n) + if _, err := rand.Read(b); err != nil { + return "", err + } + return hex.EncodeToString(b), nil +} diff --git a/pkg/staging/keygen_test.go b/pkg/staging/keygen_test.go new file mode 100644 index 0000000..1f04d37 --- /dev/null +++ b/pkg/staging/keygen_test.go @@ -0,0 +1,62 @@ +package staging + +import ( + "os" + "path/filepath" + "testing" +) + +func TestUpsertEnvVar(t *testing.T) { + cases := map[string]struct { + initial string // "" means the file does not exist yet + key string + value string + want string + }{ + "creates file when missing": { + initial: "", + key: "PAYER_ADDRESS", + value: "0xabc", + want: "PAYER_ADDRESS=0xabc\n", + }, + "appends to file ending in newline without a blank line": { + initial: "A=1\n", + key: "B", + value: "2", + want: "A=1\nB=2\n", + }, + "appends and adds EOL when input lacks trailing newline": { + initial: "A=1", + key: "B", + value: "2", + want: "A=1\nB=2\n", + }, + "replaces existing key in place": { + initial: "A=1\nB=2\n", + key: "A", + value: "9", + want: "A=9\nB=2\n", + }, + } + + for name, tc := range cases { + t.Run(name, func(t *testing.T) { + path := filepath.Join(t.TempDir(), "test.env") + if tc.initial != "" { + if err := os.WriteFile(path, []byte(tc.initial), 0o644); err != nil { + t.Fatalf("write initial file: %v", err) + } + } + if err := upsertEnvVar(path, tc.key, tc.value); err != nil { + t.Fatalf("upsertEnvVar: %v", err) + } + got, err := os.ReadFile(path) + if err != nil { + t.Fatalf("read file: %v", err) + } + if string(got) != tc.want { + t.Fatalf("got %q, want %q", string(got), tc.want) + } + }) + } +} diff --git a/pkg/staging/onepassword.go b/pkg/staging/onepassword.go new file mode 100644 index 0000000..c83e5e3 --- /dev/null +++ b/pkg/staging/onepassword.go @@ -0,0 +1,159 @@ +package staging + +import ( + "bytes" + "encoding/json" + "fmt" + "os" + "os/exec" + "sort" +) + +// opItemTemplate is the subset of 1Password's item-create JSON template we emit. +type opItemTemplate struct { + Title string `json:"title"` + Category string `json:"category"` + Fields []opField `json:"fields"` +} + +type opField struct { + ID string `json:"id"` + Type string `json:"type"` + Label string `json:"label"` + Value string `json:"value"` +} + +// storeInOnePassword writes every secret into a single 1Password item via the +// `op` CLI, using a JSON item template piped over stdin. The earlier `@` +// assignment form was a bug: `op item create/edit` does NOT expand `@file` +// (that's a curl-ism), so it stored the literal temp-file path as the value. +// A stdin template both fixes that and keeps secret values off the command line +// and out of shell history — op's own docs recommend a template for sensitive +// values, since assignment statements get logged. +// +// fields maps the 1Password field label to the temp file holding its value. +// Returns the sorted list of field labels written (names only — never values). +// +// keygen owns every field in the item: an existing item is edited in place (not +// deleted-then-recreated, which would leave nothing behind if the create +// failed), a missing one is created. keygen supplies the complete field set — +// reused values byte-for-byte as read back, fresh values for fields the item +// didn't hold yet — so the edit both refreshes existing fields (a no-op for +// reused values) and adds new ones. +// Requires an authenticated `op` session (run `op signin` first). +func storeInOnePassword(vault, item string, fields map[string]string) ([]string, error) { + if vault == "" || item == "" { + return nil, fmt.Errorf("1Password vault and item must be set") + } + if _, err := exec.LookPath("op"); err != nil { + return nil, fmt.Errorf("1Password CLI %q not found in PATH: %w", "op", err) + } + + names := make([]string, 0, len(fields)) + for label := range fields { + names = append(names, label) + } + sort.Strings(names) + + // Build the item template, reading each value from its file. Concealed fields + // stay masked in the 1Password UI; `op read op://vault/item/