Compare commits
7 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| c0ba30824e | |||
| 52e17aefbf | |||
| b6dd86fdf3 | |||
| ed91d68fbf | |||
| 531b36924c | |||
| 3a3ea74d76 | |||
| 0358efe95b |
@@ -1,21 +1,26 @@
|
||||
{
|
||||
"phase": 1,
|
||||
"phase": 13,
|
||||
"stage": "complete",
|
||||
"milestone": "v0.13",
|
||||
"milestone_slug": "production-hardening-2",
|
||||
"phase_role": "execution",
|
||||
"phase_role": "final",
|
||||
"attempts": 0,
|
||||
"updated_at": "2026-08-07T19:05:00Z",
|
||||
"milestone_complete": false,
|
||||
"updated_at": "2026-08-10T14:30:00Z",
|
||||
"milestone_complete": true,
|
||||
"previous_milestone": "v0.12",
|
||||
"phase_count": 14,
|
||||
"phases_shipped": ["P0", "P1"],
|
||||
"tags_shipped": ["v0.12.0", "v0.12.1"],
|
||||
"phases_shipped": ["P0","P1","P2","P3","P4","P5","P6","P7","P8","P9","P10","P11","P12","P13"],
|
||||
"tags_shipped": ["v0.12.0","v0.12.1","v0.12.2","v0.12.3","v0.12.4","v0.12.5","v0.12.6","v0.12.7","v0.12.8","v0.12.9","v0.12.10","v0.12.11","v0.12.12"],
|
||||
"requirements": {
|
||||
"covered": [149],
|
||||
"covered": [149,150,151,152,153,154,155,156,157,158,159,160,161,162,163],
|
||||
"partial": []
|
||||
},
|
||||
"binding_conditions": ["C-39","C-40","C-41","C-42","C-43","C-44","C-45","C-46","C-47","C-48","C-49"],
|
||||
"load_bearing_rule": "R-022",
|
||||
"next_milestone": "v1.0"
|
||||
"next_milestone": "v1.0",
|
||||
"ship": {
|
||||
"tag": "v0.12.13",
|
||||
"merged_to_milestone": true,
|
||||
"milestone_release": "v0.13"
|
||||
}
|
||||
}
|
||||
|
||||
+47
-47
@@ -254,7 +254,7 @@ operator decision Q2=C.
|
||||
|
||||
## v0.12 Milestone Summary — Security Hardening (Zero-Trust Identity)
|
||||
|
||||
**Status**: in progress (Phase 0). 30 net-new requirements (REQ-119..REQ-148)
|
||||
**Status**: complete (shipped as v0.11.x tags; milestone release v0.11.28). 30 net-new requirements (REQ-119..REQ-148)
|
||||
derived from the v0.12 threat-model review (25 findings F1..F25) and the
|
||||
zero-trust identity model (R-021). See ROADMAP.md for the 29-phase plan
|
||||
(P0 + P01..P27 + P28 final) and RESEARCH_v0.12.md for the full threat model.
|
||||
@@ -263,41 +263,41 @@ zero-trust identity model (R-021). See ROADMAP.md for the 29-phase plan
|
||||
|
||||
| ID | Requirement | Priority | Phase | Status |
|
||||
|----|-------------|----------|-------|--------|
|
||||
| REQ-119 | Command injection fix in `internal/runtime/podman.go` & `wasm.go`: shell-quote `cmdStr` via `shellQuote` in SSH exec interpolation (`podman.go:57`, `wasm.go:39`); add injection regression tests (bats + Go) covering `;`, `\|`, `$()`, backticks, newline injection (F3) | High | **v0.12 P01** | pending |
|
||||
| REQ-120 | Namespace path traversal fix: `validateNamespaceName` in `internal/ns/` rejects `..`, `/`, leading `-`, null bytes, control chars in `ns create`/`ns inherit`/`ns set-constraint`; add fuzz test (F4) | High | **v0.12 P02** | pending |
|
||||
| REQ-121 | Txn apply path allowlist: `apply.sh` python heredoc validates every `path` in `desired-state.json` against a prefix allowlist (`/etc/orca/`, `/etc/traefik/orca*`, `/etc/systemd/system/orca-*`, `/etc/nftables.d/orca*`, `/etc/syncthing/orca*`); rejects otherwise; HMAC-signed manifest unchanged (F5) | High | **v0.12 P03** | pending |
|
||||
| REQ-119 | Command injection fix in `internal/runtime/podman.go` & `wasm.go`: shell-quote `cmdStr` via `shellQuote` in SSH exec interpolation (`podman.go:57`, `wasm.go:39`); add injection regression tests (bats + Go) covering `;`, `\|`, `$()`, backticks, newline injection (F3) | High | **v0.12 P01** | complete |
|
||||
| REQ-120 | Namespace path traversal fix: `validateNamespaceName` in `internal/ns/` rejects `..`, `/`, leading `-`, null bytes, control chars in `ns create`/`ns inherit`/`ns set-constraint`; add fuzz test (F4) | High | **v0.12 P02** | complete |
|
||||
| REQ-121 | Txn apply path allowlist: `apply.sh` python heredoc validates every `path` in `desired-state.json` against a prefix allowlist (`/etc/orca/`, `/etc/traefik/orca*`, `/etc/systemd/system/orca-*`, `/etc/nftables.d/orca*`, `/etc/syncthing/orca*`); rejects otherwise; HMAC-signed manifest unchanged (F5) | High | **v0.12 P03** | complete |
|
||||
|
||||
### Wave B — Zero-trust identity
|
||||
|
||||
| ID | Requirement | Priority | Phase | Status |
|
||||
|----|-------------|----------|-------|--------|
|
||||
| REQ-122 | ACL enforcement wiring: `acl.Check` invoked in daemon handlers (read/write/admin by route) and SSH-push applier (validates `ORCA_OIDC_TOKEN` env var against JWKS before applying any txn); deny-by-default enforced; actor recorded in audit (F1, foundational for REQ-145) | High | **v0.12 P06** | pending |
|
||||
| REQ-123 | Daemon auth hardening: mandatory mTLS (remove plaintext mode entirely); OIDC bearer accepted as second factor on human-facing endpoints; `MaxBytesReader` body limits; pprof loopback-only by default, refuse non-loopback without `--pprof-allow-public` confirmation (F6, F24) | High | **v0.12 P09** | pending |
|
||||
| REQ-124 | HTTP request body size limits: `http.MaxBytesReader` on all JSON-decoding handlers; `MaxHeaderBytes` set; rejects oversized bodies (F24) | Medium | **v0.12 P09** | pending |
|
||||
| REQ-125 | Audit log tamper-evidence: hash-chained entries (`prev_hash = sha256(prev_row \|\| payload)`), HMAC-SHA256 under master key on the chain head; `orca doctor audit` verifies the chain; append-only enforcement via SQLite trigger blocking UPDATE/DELETE; actor field carries OIDC `sub` or SPIFFE SVID (F2) | High | **v0.12 P10** | pending |
|
||||
| REQ-126 | SVID chain validation: `VerifySVID` validates the full cert chain against the CA pool, not just the URI SAN; reject certs signed by unknown CAs even with correct URI (F9) | High | **v0.12 P11** | pending |
|
||||
| REQ-127 | Backup symlink validation: `Restore` rejects `Linkname` that's absolute, contains `..`, or points outside `ORCA_HOME`; add regression test with crafted tarball (F7) | High | **v0.12 P12** | pending |
|
||||
| REQ-128 | step-ca /tmp hardening: `step ca certificate` writes to 0600 temp under `ClusterDir()/step-tmp/` (or `TMPDIR` override), not world-readable `/tmp`; cleanup in `defer` (F10) | High | **v0.12 P13** | pending |
|
||||
| REQ-129 | Master key rotation: `orca secrets rotate-master` re-encrypts all namespace secrets under a new master key; new master key re-sealed to OIDC as part of the same operation; `--dry-run` + atomic + automatic rollback to old sealed key on any ns failure; no passphrase (R-021) (F12) | High | **v0.12 P14** | pending |
|
||||
| REQ-130 | File-mode audit expansion: `EnforceFileModes` extended to SSH key, master key (sealed blob), server cert/key, known_hosts; `orca doctor modes` checks all; startup refuses to run on violation (F13) | Medium | **v0.12 P15** | pending |
|
||||
| REQ-131 | aggregate.sh JSON injection fix + drift-gate parse fix: replace `printf` interpolation with `jq`-based JSON construction (or Go-side aggregator emitting JSON); fix `orca-pull.sh` R-020 parsing to use `jq` instead of grep (F11, F18) | High | **v0.12 P16** | pending |
|
||||
| REQ-132 | install.sh checksum+GPG verification: release.sh publishes `SHA256SUMS` + `SHA256SUMS.asc` (GPG-signed) alongside tarball; install.sh verifies before `tar -xzf`; fail closed on mismatch (F14) | High | **v0.12 P17** | pending |
|
||||
| REQ-133 | nftables ruleset hardening: add conntrack bounds (`ct state established,related accept`), input default-deny on orca chain, drop invalid packets; `orca doctor nft` audits live ruleset against emitted one (F21) | Medium | **v0.12 P18** | pending |
|
||||
| REQ-134 | sudoers hardening: add NOEXEC to `apt-get`/`dpkg` (or remove if unused); `orca doctor proxmox` audits sudoers file against expected allowlist (F22) | Medium | **v0.12 P19** | pending |
|
||||
| REQ-135 | System user consistency: Proxmox bootstrap creates `nologin` system user (`-r -s /usr/sbin/nologin`), matching peer-setup; `orca doctor` flags inconsistency on existing peers; `orca upgrade` migrates (F23) | Medium | **v0.12 P20** | pending |
|
||||
| REQ-136 | SQLite file-mode + at-rest encryption: `store.Open` sets DB file mode 0600; optional `--encrypt-db` (CGO-free fallback per C-31: file-mode 0600 + documented threat if SQLCipher needs CGO); no CGO (F8) | High | **v0.12 P21** | pending |
|
||||
| REQ-137 | Migration safety: `copyFile` -> atomic temp+rename; `migrateDBSchema` runs in transaction with `foreign_keys(ON)`; pre-migration backup step (uses `internal/backup`); document manual rollback; v0.11->v0.12 identity migration: `orca upgrade` refuses clusters using `--password`/bare-tokens without `--accept-identity-migration` (F19, C-34) | High | **v0.12 P22** | pending |
|
||||
| REQ-138 | Legacy CA/mTLS/daemon + step-ca password-provisioner deletion: remove `internal/security/ca.go` legacy CA, `internal/transport/mtls.go` deprecated path, daemon plaintext mode; migrate `orca init`/`orca cert *` to step-ca exclusively; `certpaths` (v0.8 layout) removed; delete step-ca `--password-file` provisioner (replaced by OIDC provisioner); **gate: P06/P08/P09/P11 all shipped** (F16) | High | **v0.12 P23** | pending |
|
||||
| REQ-139 | known_hosts tightening + transport hardening: `Flock` tightens pre-existing looser perms to 0600; `classifyDialErr` switched from substring to typed errors; add SSH-exec rate limiting (token bucket per peer) (F15, F25) | Medium | **v0.12 P24** | pending |
|
||||
| REQ-140 | Drift event authentication: drift events signed with per-peer HMAC key (derived from master key); aggregator rejects unsigned/forged events; `orca-drift-notify.sh` reads key from 0600 file owned by `orca` (F18) | Medium | **v0.12 P25** | pending |
|
||||
| REQ-141 | Security integration test suite: hermetic harness exercising injection, traversal, symlink, drift-forgery, audit-tamper, daemon-auth-negative, OIDC mock-IdP flow, ACL-with-OIDC-claims negative tests, unseal/seal, WebAuthn virtual-authenticator ceremony, password-removal regression (assert `--password` is rejected); gates in `.coreci.yml` `validate` (C-33) | High | **v0.12 P26** | pending |
|
||||
| REQ-142 | Zero-trust + OIDC + WebAuthn + threat-model docs: `docs/threat-model.md` (STRIDE + zero-trust model + OIDC data-flow), `docs/oidc.md` (configure your IdP, Dex offline quickstart, claim-to-namespace mapping), `docs/webauthn.md` (passkey registration, RP ID, secure context), `docs/security-runbook.md` (unseal/seal, master key rotation, incident response, sudoers audit, nft audit); README security section names "no orca credentials" as an invariant | Medium | **v0.12 P27** | pending |
|
||||
| REQ-143 | Final review + ship + audit: multi-persona review across all phases, `ciagent-audit` reconstruction test, milestone merge to main, tag `v0.11.29` (= v0.12 milestone release per feature-milestone rule) | High | **v0.12 P28** | pending |
|
||||
| REQ-144 | OIDC client + bundled Dex: `orca auth login`/`logout`/`status`/`init-idp`; OIDC config block (`oidc.issuer`, `client_id`, `client_secret`, `scopes`); bundled Dex systemd unit + Traefik route on the lead; BYO external IdP override via `oidc.issuer` repoint; JWKS caching + refresh; token storage at `~/.orca/credentials.json` (0600); `--oidc` flag on commands requiring identity; browser auth-code + PKCE + local loopback redirect; headless device-code fallback (D-238..D-247) | High | **v0.12 P04** | pending |
|
||||
| REQ-145 | ACL rewrite to OIDC claims: remove `KindToken` entirely; `KindSpiffe` stays for machine identity; new `KindOidc` maps `sub`+`groups` -> namespace permissions; `acl.Check` takes OIDC claims struct; deny-by-default enforced in daemon + SSH-push applier; `acl.json` mode tightened to 0600 (F1) | High | **v0.12 P06** | pending |
|
||||
| REQ-146 | Remove all password/token paths (breaking): delete `--password`/`$ORCA_PROXMOX_PASSWORD` from Proxmox join (replace with pre-staged-key-only or `step ssh` OIDC cert exchange); delete step-ca `--password-file` provisioner (migrate to OIDC provisioner); delete any bare-token CLI paths; documented in migration guide (R-021, C-34) | High | **v0.12 P07** | pending |
|
||||
| REQ-147 | Master key seal-to-OIDC + Shamir recovery: master key encrypted with key derived from OIDC token exchange at unseal; `orca cluster unseal`/`seal`; sealed blob at `ClusterDir()/master.key.sealed` (0600); raw key never on disk; Shamir 3-of-5 shards printed at seal time; recovery via `--recovery` + 3 shards; mTLS-only offline path derives seal key from cluster CA (D-241, C-35) | High | **v0.12 P08** | pending |
|
||||
| REQ-148 | WebAuthn connector for Dex (passkeys): `orca-webauthn-connector` (~300 LoC Go, `go-webauthn`); register/login ceremonies at `/orca/webauthn/{register,login}` behind Traefik; `orca auth register` browser flow; passkey storage SQLite `ClusterDir()/webauthn-credentials.db` (0600, public keys only); RP ID = cluster Traefik domain; secure context via step-ca cert; headless device-code fallback; virtual-authenticator integration tests (D-240, D-243, D-244, C-38) | High | **v0.12 P05** | pending |
|
||||
| REQ-122 | ACL enforcement wiring: `acl.Check` invoked in daemon handlers (read/write/admin by route) and SSH-push applier (validates `ORCA_OIDC_TOKEN` env var against JWKS before applying any txn); deny-by-default enforced; actor recorded in audit (F1, foundational for REQ-145) | High | **v0.12 P06** | complete |
|
||||
| REQ-123 | Daemon auth hardening: mandatory mTLS (remove plaintext mode entirely); OIDC bearer accepted as second factor on human-facing endpoints; `MaxBytesReader` body limits; pprof loopback-only by default, refuse non-loopback without `--pprof-allow-public` confirmation (F6, F24) | High | **v0.12 P09** | complete |
|
||||
| REQ-124 | HTTP request body size limits: `http.MaxBytesReader` on all JSON-decoding handlers; `MaxHeaderBytes` set; rejects oversized bodies (F24) | Medium | **v0.12 P09** | complete |
|
||||
| REQ-125 | Audit log tamper-evidence: hash-chained entries (`prev_hash = sha256(prev_row \|\| payload)`), HMAC-SHA256 under master key on the chain head; `orca doctor audit` verifies the chain; append-only enforcement via SQLite trigger blocking UPDATE/DELETE; actor field carries OIDC `sub` or SPIFFE SVID (F2) | High | **v0.12 P10** | complete |
|
||||
| REQ-126 | SVID chain validation: `VerifySVID` validates the full cert chain against the CA pool, not just the URI SAN; reject certs signed by unknown CAs even with correct URI (F9) | High | **v0.12 P11** | complete |
|
||||
| REQ-127 | Backup symlink validation: `Restore` rejects `Linkname` that's absolute, contains `..`, or points outside `ORCA_HOME`; add regression test with crafted tarball (F7) | High | **v0.12 P12** | complete |
|
||||
| REQ-128 | step-ca /tmp hardening: `step ca certificate` writes to 0600 temp under `ClusterDir()/step-tmp/` (or `TMPDIR` override), not world-readable `/tmp`; cleanup in `defer` (F10) | High | **v0.12 P13** | complete |
|
||||
| REQ-129 | Master key rotation: `orca secrets rotate-master` re-encrypts all namespace secrets under a new master key; new master key re-sealed to OIDC as part of the same operation; `--dry-run` + atomic + automatic rollback to old sealed key on any ns failure; no passphrase (R-021) (F12) | High | **v0.12 P14** | complete |
|
||||
| REQ-130 | File-mode audit expansion: `EnforceFileModes` extended to SSH key, master key (sealed blob), server cert/key, known_hosts; `orca doctor modes` checks all; startup refuses to run on violation (F13) | Medium | **v0.12 P15** | complete |
|
||||
| REQ-131 | aggregate.sh JSON injection fix + drift-gate parse fix: replace `printf` interpolation with `jq`-based JSON construction (or Go-side aggregator emitting JSON); fix `orca-pull.sh` R-020 parsing to use `jq` instead of grep (F11, F18) | High | **v0.12 P16** | complete |
|
||||
| REQ-132 | install.sh checksum+GPG verification: release.sh publishes `SHA256SUMS` + `SHA256SUMS.asc` (GPG-signed) alongside tarball; install.sh verifies before `tar -xzf`; fail closed on mismatch (F14) | High | **v0.12 P17** | complete |
|
||||
| REQ-133 | nftables ruleset hardening: add conntrack bounds (`ct state established,related accept`), input default-deny on orca chain, drop invalid packets; `orca doctor nft` audits live ruleset against emitted one (F21) | Medium | **v0.12 P18** | complete |
|
||||
| REQ-134 | sudoers hardening: add NOEXEC to `apt-get`/`dpkg` (or remove if unused); `orca doctor proxmox` audits sudoers file against expected allowlist (F22) | Medium | **v0.12 P19** | complete |
|
||||
| REQ-135 | System user consistency: Proxmox bootstrap creates `nologin` system user (`-r -s /usr/sbin/nologin`), matching peer-setup; `orca doctor` flags inconsistency on existing peers; `orca upgrade` migrates (F23) | Medium | **v0.12 P20** | complete |
|
||||
| REQ-136 | SQLite file-mode + at-rest encryption: `store.Open` sets DB file mode 0600; optional `--encrypt-db` (CGO-free fallback per C-31: file-mode 0600 + documented threat if SQLCipher needs CGO); no CGO (F8) | High | **v0.12 P21** | complete |
|
||||
| REQ-137 | Migration safety: `copyFile` -> atomic temp+rename; `migrateDBSchema` runs in transaction with `foreign_keys(ON)`; pre-migration backup step (uses `internal/backup`); document manual rollback; v0.11->v0.12 identity migration: `orca upgrade` refuses clusters using `--password`/bare-tokens without `--accept-identity-migration` (F19, C-34) | High | **v0.12 P22** | complete |
|
||||
| REQ-138 | Legacy CA/mTLS/daemon + step-ca password-provisioner deletion: remove `internal/security/ca.go` legacy CA, `internal/transport/mtls.go` deprecated path, daemon plaintext mode; migrate `orca init`/`orca cert *` to step-ca exclusively; `certpaths` (v0.8 layout) removed; delete step-ca `--password-file` provisioner (replaced by OIDC provisioner); **gate: P06/P08/P09/P11 all shipped** (F16) | High | **v0.12 P23** | complete |
|
||||
| REQ-139 | known_hosts tightening + transport hardening: `Flock` tightens pre-existing looser perms to 0600; `classifyDialErr` switched from substring to typed errors; add SSH-exec rate limiting (token bucket per peer) (F15, F25) | Medium | **v0.12 P24** | complete |
|
||||
| REQ-140 | Drift event authentication: drift events signed with per-peer HMAC key (derived from master key); aggregator rejects unsigned/forged events; `orca-drift-notify.sh` reads key from 0600 file owned by `orca` (F18) | Medium | **v0.12 P25** | complete |
|
||||
| REQ-141 | Security integration test suite: hermetic harness exercising injection, traversal, symlink, drift-forgery, audit-tamper, daemon-auth-negative, OIDC mock-IdP flow, ACL-with-OIDC-claims negative tests, unseal/seal, WebAuthn virtual-authenticator ceremony, password-removal regression (assert `--password` is rejected); gates in `.coreci.yml` `validate` (C-33) | High | **v0.12 P26** | complete |
|
||||
| REQ-142 | Zero-trust + OIDC + WebAuthn + threat-model docs: `docs/threat-model.md` (STRIDE + zero-trust model + OIDC data-flow), `docs/oidc.md` (configure your IdP, Dex offline quickstart, claim-to-namespace mapping), `docs/webauthn.md` (passkey registration, RP ID, secure context), `docs/security-runbook.md` (unseal/seal, master key rotation, incident response, sudoers audit, nft audit); README security section names "no orca credentials" as an invariant | Medium | **v0.12 P27** | complete |
|
||||
| REQ-143 | Final review + ship + audit: multi-persona review across all phases, `ciagent-audit` reconstruction test, milestone merge to main, tag `v0.11.29` (= v0.12 milestone release per feature-milestone rule) | High | **v0.12 P28** | complete |
|
||||
| REQ-144 | OIDC client + bundled Dex: `orca auth login`/`logout`/`status`/`init-idp`; OIDC config block (`oidc.issuer`, `client_id`, `client_secret`, `scopes`); bundled Dex systemd unit + Traefik route on the lead; BYO external IdP override via `oidc.issuer` repoint; JWKS caching + refresh; token storage at `~/.orca/credentials.json` (0600); `--oidc` flag on commands requiring identity; browser auth-code + PKCE + local loopback redirect; headless device-code fallback (D-238..D-247) | High | **v0.12 P04** | complete |
|
||||
| REQ-145 | ACL rewrite to OIDC claims: remove `KindToken` entirely; `KindSpiffe` stays for machine identity; new `KindOidc` maps `sub`+`groups` -> namespace permissions; `acl.Check` takes OIDC claims struct; deny-by-default enforced in daemon + SSH-push applier; `acl.json` mode tightened to 0600 (F1) | High | **v0.12 P06** | complete |
|
||||
| REQ-146 | Remove all password/token paths (breaking): delete `--password`/`$ORCA_PROXMOX_PASSWORD` from Proxmox join (replace with pre-staged-key-only or `step ssh` OIDC cert exchange); delete step-ca `--password-file` provisioner (migrate to OIDC provisioner); delete any bare-token CLI paths; documented in migration guide (R-021, C-34) | High | **v0.12 P07** | complete |
|
||||
| REQ-147 | Master key seal-to-OIDC + Shamir recovery: master key encrypted with key derived from OIDC token exchange at unseal; `orca cluster unseal`/`seal`; sealed blob at `ClusterDir()/master.key.sealed` (0600); raw key never on disk; Shamir 3-of-5 shards printed at seal time; recovery via `--recovery` + 3 shards; mTLS-only offline path derives seal key from cluster CA (D-241, C-35) | High | **v0.12 P08** | complete |
|
||||
| REQ-148 | WebAuthn connector for Dex (passkeys): `orca-webauthn-connector` (~300 LoC Go, `go-webauthn`); register/login ceremonies at `/orca/webauthn/{register,login}` behind Traefik; `orca auth register` browser flow; passkey storage SQLite `ClusterDir()/webauthn-credentials.db` (0600, public keys only); RP ID = cluster Traefik domain; secure context via step-ca cert; headless device-code fallback; virtual-authenticator integration tests (D-240, D-243, D-244, C-38) | High | **v0.12 P05** | complete |
|
||||
|
||||
### Scope notes (v0.12)
|
||||
|
||||
@@ -311,7 +311,7 @@ zero-trust identity model (R-021). See ROADMAP.md for the 29-phase plan
|
||||
|
||||
## Milestone v0.13: Production Hardening Round 2 + UAT Plan
|
||||
|
||||
**Status**: in progress (2026-08-07). v0.12 (Security Hardening) is
|
||||
**Status**: complete (2026-08-10). v0.12 (Security Hardening) is
|
||||
COMPLETE; v0.13 is the final hardening round before the v1.0.0
|
||||
production-ready tag. v1.0.0 is gated on the UAT signoff script
|
||||
(`scripts/uat-signoff.sh`) delivered by this milestone.
|
||||
@@ -320,41 +320,41 @@ production-ready tag. v1.0.0 is gated on the UAT signoff script
|
||||
|
||||
| ID | Requirement | Priority | Phase | Status |
|
||||
|----|-------------|----------|-------|--------|
|
||||
| REQ-149 | Go toolchain bump to 1.25.12+ (closes 24 stdlib vulns: archive/tar GO-2025-4014/GO-2026-4869, crypto/tls GO-2026-5856/GO-2025-4008, crypto/x509 GO-2026-5037/4947/4946/GO-2025-4175/4155/4013, net/http GO-2026-4918/GO-2025-4012, net/url GO-2026-4601/4341/GO-2025-4010, encoding/pem GO-2025-4009, os GO-2026-4602); `govulncheck -show verbose` triage of 6 imported third-party vulns; bump deps with reachable traces | High | **v0.13 P01** | pending |
|
||||
| REQ-150 | Input validation & injection hardening: (a) `orca logs --job` validate against `^[A-Za-z0-9_-]+$`, use `shellQuote` not `%q` (critical: backtick RCE via SSH fanout); (b) pprof `isLoopback(":6060")` treat empty host as non-loopback/bind-all, reject unless explicit public-allow flag wired; remove phantom `--pprof-allow-public` references, make loopback-only a hard invariant; (c) backup restore tar-slip fix: use `filepath.Rel(target, dest)` containment check instead of `HasPrefix(name, "..")`; (d) `orca txn rollback` validate txn ID against `^T-[0-9a-f]{16}$`; (e) `orca nft diff --against` validate txn ID before `filepath.Join`; (f) `drain stopAlloc` validate `allocID` against `^[A-Za-z0-9_-]+$` before `systemctl stop`; (g) `cluster_compat` `shellQuote(first)` for peer dir name; (h) `runtime/podman.go` use `shellQuote(image)` not `%q`; (i) nft `TrustedProbes` validate each entry with `net.ParseIP`/`net.ParseCIDR`; (j) sudoers: validate `--proxmox-user`/`--proxmox-role` against `^[a-z_][a-z0-9_-]{0,31}$`; write to fixed `/etc/sudoers.d/orca`; `shellQuote` all pveum/useradd; `validateSudoers` check the actual file written; (k) `nft country block add` validate `^[A-Z]{2}$` | Critical | **v0.13 P02** | pending |
|
||||
| REQ-149 | Go toolchain bump to 1.25.12+ (closes 24 stdlib vulns: archive/tar GO-2025-4014/GO-2026-4869, crypto/tls GO-2026-5856/GO-2025-4008, crypto/x509 GO-2026-5037/4947/4946/GO-2025-4175/4155/4013, net/http GO-2026-4918/GO-2025-4012, net/url GO-2026-4601/4341/GO-2025-4010, encoding/pem GO-2025-4009, os GO-2026-4602); `govulncheck -show verbose` triage of 6 imported third-party vulns; bump deps with reachable traces | High | **v0.13 P01** | complete |
|
||||
| REQ-150 | Input validation & injection hardening: (a) `orca logs --job` validate against `^[A-Za-z0-9_-]+$`, use `shellQuote` not `%q` (critical: backtick RCE via SSH fanout); (b) pprof `isLoopback(":6060")` treat empty host as non-loopback/bind-all, reject unless explicit public-allow flag wired; remove phantom `--pprof-allow-public` references, make loopback-only a hard invariant; (c) backup restore tar-slip fix: use `filepath.Rel(target, dest)` containment check instead of `HasPrefix(name, "..")`; (d) `orca txn rollback` validate txn ID against `^T-[0-9a-f]{16}$`; (e) `orca nft diff --against` validate txn ID before `filepath.Join`; (f) `drain stopAlloc` validate `allocID` against `^[A-Za-z0-9_-]+$` before `systemctl stop`; (g) `cluster_compat` `shellQuote(first)` for peer dir name; (h) `runtime/podman.go` use `shellQuote(image)` not `%q`; (i) nft `TrustedProbes` validate each entry with `net.ParseIP`/`net.ParseCIDR`; (j) sudoers: validate `--proxmox-user`/`--proxmox-role` against `^[a-z_][a-z0-9_-]{0,31}$`; write to fixed `/etc/sudoers.d/orca`; `shellQuote` all pveum/useradd; `validateSudoers` check the actual file written; (k) `nft country block add` validate `^[A-Z]{2}$` | Critical | **v0.13 P02** | complete |
|
||||
|
||||
### Wave B — Scheduler wiring & jobspec parser (architectural)
|
||||
|
||||
| ID | Requirement | Priority | Phase | Status |
|
||||
|----|-------------|----------|-------|--------|
|
||||
| REQ-151 | Scheduler/deployment wiring: wire `internal/scheduler.Schedule()` into `orca job run` — replace local `exec.CommandContext` path with: evaluate constraints/capacity/affinity via scheduler → render systemd units via `internal/emitter` → SSH-push to target via `internal/sshpush`; `--target` overrides scheduler selection; capacity enforced (reject job if no node fits); CEL constraints evaluated; affinity weighted scoring; `systemd-analyze verify` on rendered unit before deploy; `job run` without `--target` uses scheduler bin-packing across registered nodes | Critical | **v0.13 P03** | pending |
|
||||
| REQ-152 | jobspec parser fixes: add `case "schedule":` and `case "timeout":` to top-level switch in `internal/jobspec/markdown.go` (currently silently dropped); fix DaemonSet — parser must not default Count to 1 for DaemonSet (validator rejects Count!=0); DaemonSet schedule block actually parsed and stored; `timeout:` on Jobs parsed and enforced (kill after duration); `restart:` policy translated to systemd `Restart=`/`StartLimitBurst` in emitter; add `job lint` warnings for advisory-only fields (cron, health, update, affinity) with honest "not enforced in this version" message | Critical | **v0.13 P03** | pending |
|
||||
| REQ-151 | Scheduler/deployment wiring: wire `internal/scheduler.Schedule()` into `orca job run` — replace local `exec.CommandContext` path with: evaluate constraints/capacity/affinity via scheduler → render systemd units via `internal/emitter` → SSH-push to target via `internal/sshpush`; `--target` overrides scheduler selection; capacity enforced (reject job if no node fits); CEL constraints evaluated; affinity weighted scoring; `systemd-analyze verify` on rendered unit before deploy; `job run` without `--target` uses scheduler bin-packing across registered nodes | Critical | **v0.13 P03** | complete |
|
||||
| REQ-152 | jobspec parser fixes: add `case "schedule":` and `case "timeout":` to top-level switch in `internal/jobspec/markdown.go` (currently silently dropped); fix DaemonSet — parser must not default Count to 1 for DaemonSet (validator rejects Count!=0); DaemonSet schedule block actually parsed and stored; `timeout:` on Jobs parsed and enforced (kill after duration); `restart:` policy translated to systemd `Restart=`/`StartLimitBurst` in emitter; add `job lint` warnings for advisory-only fields (cron, health, update, affinity) with honest "not enforced in this version" message | Critical | **v0.13 P03** | complete |
|
||||
|
||||
### Wave C — Zero-trust enforcement wiring
|
||||
|
||||
| ID | Requirement | Priority | Phase | Status |
|
||||
|----|-------------|----------|-------|--------|
|
||||
| REQ-153 | ACL enforcement + WebAuthn registration auth: (a) wire `acl.Check` into all 5 daemon handlers (`dispatch`/`jobs`/`nodes`/`tasks`/`health`) — extract OIDC sub/SPIFFE SVID from mTLS peer cert, check against ACL for namespace+verb, deny-by-default; (b) wire `acl.Check` into sshpush applier + txn apply path (validate `ORCA_OIDC_TOKEN` bearer against JWKS); (c) thread OIDC sub/SVID into audit `actor` field (replaces "cli"/"daemon"); (d) fix `acl.json` mode 0644→0600; (e) fix WebAuthn unauthenticated registration — `/orca/webauthn/register` requires existing authenticated session or admin bootstrap token; do not allow overwriting existing credentials without re-auth; (f) add flock on `acl.json` for concurrent grant/revoke | Critical | **v0.13 P04** | pending |
|
||||
| REQ-154 | Seal/audit CLI + chain race + key zeroing: (a) implement `orca cluster seal`/`unseal` (OIDC token exchange→unwrap master key→zeroed on shutdown; Shamir 3-of-5 shards printed at seal time; sealed blob at `ClusterDir()/master.key.sealed` 0600); (b) implement `orca doctor audit` (invokes `AuditRepo.VerifyChain`); (c) implement `orca doctor modes` (invokes `EnforceFileModes` across ORCA_HOME); (d) fix audit hash-chain race — `Append` uses `BEGIN IMMEDIATE` transaction; (e) fix `secrets rotate-master` to actually re-seal to OIDC; (f) zero master key / namespace keys / SVID private keys after use (defense-in-depth against pprof heap extraction) | High | **v0.13 P05** | pending |
|
||||
| REQ-155 | auth init-idp real + auth register: (a) implement `orca auth init-idp` — render Dex systemd unit + config template + Traefik dynamic route from `internal/webauthn/` connector at `https://<cluster>/orca/webauthn/{register,login}`; RP ID = cluster Traefik domain (C-38); HTTPS secure context via step-ca cert; atomic deploy with rollback; (b) implement `orca auth register` (browser flow to WebAuthn registration endpoint); (c) `loadOIDCConfig` config-file loading (`oidc.issuer` in config, not flags-only); (d) `orca doctor oidc` health check | High | **v0.13 P06** | pending |
|
||||
| REQ-153 | ACL enforcement + WebAuthn registration auth: (a) wire `acl.Check` into all 5 daemon handlers (`dispatch`/`jobs`/`nodes`/`tasks`/`health`) — extract OIDC sub/SPIFFE SVID from mTLS peer cert, check against ACL for namespace+verb, deny-by-default; (b) wire `acl.Check` into sshpush applier + txn apply path (validate `ORCA_OIDC_TOKEN` bearer against JWKS); (c) thread OIDC sub/SVID into audit `actor` field (replaces "cli"/"daemon"); (d) fix `acl.json` mode 0644→0600; (e) fix WebAuthn unauthenticated registration — `/orca/webauthn/register` requires existing authenticated session or admin bootstrap token; do not allow overwriting existing credentials without re-auth; (f) add flock on `acl.json` for concurrent grant/revoke | Critical | **v0.13 P04** | complete |
|
||||
| REQ-154 | Seal/audit CLI + chain race + key zeroing: (a) implement `orca cluster seal`/`unseal` (OIDC token exchange→unwrap master key→zeroed on shutdown; Shamir 3-of-5 shards printed at seal time; sealed blob at `ClusterDir()/master.key.sealed` 0600); (b) implement `orca doctor audit` (invokes `AuditRepo.VerifyChain`); (c) implement `orca doctor modes` (invokes `EnforceFileModes` across ORCA_HOME); (d) fix audit hash-chain race — `Append` uses `BEGIN IMMEDIATE` transaction; (e) fix `secrets rotate-master` to actually re-seal to OIDC; (f) zero master key / namespace keys / SVID private keys after use (defense-in-depth against pprof heap extraction) | High | **v0.13 P05** | complete |
|
||||
| REQ-155 | auth init-idp real + auth register: (a) implement `orca auth init-idp` — render Dex systemd unit + config template + Traefik dynamic route from `internal/webauthn/` connector at `https://<cluster>/orca/webauthn/{register,login}`; RP ID = cluster Traefik domain (C-38); HTTPS secure context via step-ca cert; atomic deploy with rollback; (b) implement `orca auth register` (browser flow to WebAuthn registration endpoint); (c) `loadOIDCConfig` config-file loading (`oidc.issuer` in config, not flags-only); (d) `orca doctor oidc` health check | High | **v0.13 P06** | complete |
|
||||
|
||||
### Wave D — Concurrency, transport, migration safety
|
||||
|
||||
| ID | Requirement | Priority | Phase | Status |
|
||||
|----|-------------|----------|-------|--------|
|
||||
| REQ-156 | Concurrency safety: (a) SQLite `busy_timeout(5000)` + `SetMaxOpenConns(1)` on all DSNs (store, cache, recovery, webauthn); (b) secrets file flock (concurrent `secrets set` on same ns no longer loses data); (c) upgrade lock file (refuse concurrent `orca upgrade`); (d) backup lock file; (e) cache invalidation by write commands (`node join`/`leave`, `ns create`/`delete`, `job run`/`stop` invalidate relevant cache class — read-after-write consistency); (f) `Executor.Run` mutex scope fix (hold only for DB inserts, not whole job duration); (g) `ns create` atomic dir+ns.md write; (h) `writeCurrentLead` atomic write; (i) consolidate 3 divergent `writeAtomic` impls onto `security.WriteAtomic`; (j) WebAuthn session stores guarded with `sync.Mutex` | High | **v0.13 P07** | pending |
|
||||
| REQ-157 | Transport & SSH safety: (a) replace substring matching in `transport.IsTransient` AND `sshpush.isTransient` with typed sentinels (`errors.Is`); (b) `rotateSSHKeys` 2-phase atomic swap (stage new key on all peers → atomic swap → verify → cleanup old); (c) `known_hosts` flock field actually read by `dial()` (TOFU callback uses new field, not v0.8 `certpaths.KnownHostsPath()`); (d) IPv6 `net.JoinHostPort` in proxmox SSH dial + drain `splitHostPort`; (e) explicit timeouts for all SSH commands (peer-setup, drift remediate/ack, txn rollback, job restart — use `context.WithTimeout`); (f) `verifyCutover` use `security.ClientTLSConfig` with orca CA pool; (g) OIDC callback server `ReadHeaderTimeout: 5s`; (h) root SIGINT/SIGTERM handler for non-watch commands (clean SSH session + temp file cleanup) | High | **v0.13 P08** | pending |
|
||||
| REQ-158 | Migration & operational safety: (a) migration transaction + torn-write fix — `migrateDBSchema` wraps ALTER TABLE in transaction; crash after `os.Rename` but before schema fixup is recoverable; (b) `job stop` real `systemctl stop` via SSH (matches `job restart` pattern; honest semantics); (c) DB retention/compaction for `jobs`/`tasks`/`audit_log` tables (retention policy + `orca doctor db` compaction check); (d) `orca logs --lines` cap + `--since` upper bound (prevent OOM from unbounded journalctl output); (e) cache DB mode 0600 (matches `store.Open`); (f) `upgrade.go` cutover backup-file + atomic-rename (replace direct `sed -i`) | High | **v0.13 P09** | pending |
|
||||
| REQ-156 | Concurrency safety: (a) SQLite `busy_timeout(5000)` + `SetMaxOpenConns(1)` on all DSNs (store, cache, recovery, webauthn); (b) secrets file flock (concurrent `secrets set` on same ns no longer loses data); (c) upgrade lock file (refuse concurrent `orca upgrade`); (d) backup lock file; (e) cache invalidation by write commands (`node join`/`leave`, `ns create`/`delete`, `job run`/`stop` invalidate relevant cache class — read-after-write consistency); (f) `Executor.Run` mutex scope fix (hold only for DB inserts, not whole job duration); (g) `ns create` atomic dir+ns.md write; (h) `writeCurrentLead` atomic write; (i) consolidate 3 divergent `writeAtomic` impls onto `security.WriteAtomic`; (j) WebAuthn session stores guarded with `sync.Mutex` | High | **v0.13 P07** | complete |
|
||||
| REQ-157 | Transport & SSH safety: (a) replace substring matching in `transport.IsTransient` AND `sshpush.isTransient` with typed sentinels (`errors.Is`); (b) `rotateSSHKeys` 2-phase atomic swap (stage new key on all peers → atomic swap → verify → cleanup old); (c) `known_hosts` flock field actually read by `dial()` (TOFU callback uses new field, not v0.8 `certpaths.KnownHostsPath()`); (d) IPv6 `net.JoinHostPort` in proxmox SSH dial + drain `splitHostPort`; (e) explicit timeouts for all SSH commands (peer-setup, drift remediate/ack, txn rollback, job restart — use `context.WithTimeout`); (f) `verifyCutover` use `security.ClientTLSConfig` with orca CA pool; (g) OIDC callback server `ReadHeaderTimeout: 5s`; (h) root SIGINT/SIGTERM handler for non-watch commands (clean SSH session + temp file cleanup) | High | **v0.13 P08** | complete |
|
||||
| REQ-158 | Migration & operational safety: (a) migration transaction + torn-write fix — `migrateDBSchema` wraps ALTER TABLE in transaction; crash after `os.Rename` but before schema fixup is recoverable; (b) `job stop` real `systemctl stop` via SSH (matches `job restart` pattern; honest semantics); (c) DB retention/compaction for `jobs`/`tasks`/`audit_log` tables (retention policy + `orca doctor db` compaction check); (d) `orca logs --lines` cap + `--since` upper bound (prevent OOM from unbounded journalctl output); (e) cache DB mode 0600 (matches `store.Open`); (f) `upgrade.go` cutover backup-file + atomic-rename (replace direct `sed -i`) | High | **v0.13 P09** | complete |
|
||||
|
||||
### Wave E — Observability, docs, UAT
|
||||
|
||||
| ID | Requirement | Priority | Phase | Status |
|
||||
|----|-------------|----------|-------|--------|
|
||||
| REQ-159 | Observability expansion: metrics add `orca_jobs_by_state` histogram, `orca_drift_events_total` counter, `orca_ssh_errors_total` counter, `orca_txn_apply_total`/`orca_txn_rollback_total` counters, `orca_acl_denials_total` counter, `orca_audit_chain_head` gauge; new `docs/metrics.md` with Prometheus scrape config; security headers middleware on daemon (`X-Content-Type-Options`, `X-Frame-Options`) | Medium | **v0.13 P10** | pending |
|
||||
| REQ-160 | Doc drift round 2: (a) README — update status banner (v0.12+v0.13 complete), latest tag, subcommand table (add `auth`/`nft`/`peer-setup`/`secrets rotate-master`), correct "mTLS by default" claim (SSH-push is canonical, mTLS deprecated), add missing docs to table; (b) `docs/cli.md` — complete rewrite covering all ~40 subcommands; (c) CHANGELOG regen; (d) help text fixes (`job run` HCL→markdown, `job stop` daemon→SSH-push); (e) `docs/webauthn.md` add `auth register`; (f) `docs/namespace.md` add `inherit`/`set-constraint`; (g) `docs/install.md`+`docker.md` update version refs; (h) `docs/security-runbook.md` match P05 reality; (i) fix `verify-reqs` bold-format regex (currently bypasses v0.12); (j) fix ROADMAP/REQUIREMENTS v0.12 status hygiene; (k) `docs/security-scanning.md` gosec.json; (l) `internal/proxmox/bootstrap.go` comments (password→key auth); (m) deprecate `orca status` stub; (n) `make verify-docs` target (cli.md ↔ `orca --help` consistency) | High | **v0.13 P11** | pending |
|
||||
| REQ-161 | `--type linux` SSH-join: implement `NodeKindLinux` path (reserved at `model/node.go:29`); new `internal/linux/bootstrap.go` mirroring Proxmox pattern — orca pubkey deploy → `orca` system user → drift-events dir → no PVE role; key-auth only (R-021); `orca node join --type linux --host <ip> --ssh-user root --ssh-key <path>`; `peer-setup.go` kept as documented fallback | High | **v0.13 P12** | pending |
|
||||
| REQ-162 | UAT plan: `docs/uat.md` — 3-host topology (lead Ubuntu 22.04 + pve01 Proxmox VE 8/9 + worker01 Ubuntu 22.04); step-by-step with exact commands (bootstrap→onboard Proxmox→onboard Ubuntu worker→capacity→namespace→deploy full stack→migrate between hosts→exercise every claim); claim matrix mapping ~35 feature claims to UAT steps; signoff procedure (run `scripts/uat-signoff.sh`, paste output) | Critical | **v0.13 P12** | pending |
|
||||
| REQ-163 | UAT signoff script: `scripts/uat-signoff.sh` — idempotent, `set -euo pipefail`, ~35 named assertions covering all feature claims; read + non-mutating only (doctor, list, --dry-run); exit 0 iff all pass; `scripts/uat-smoke.sh` — pure-CLI subset for CI `validate` (version, acl file mode, doctor modes, no-password grep, metrics shape); tests for both scripts | Critical | **v0.13 P12** | pending |
|
||||
| REQ-159 | Observability expansion: metrics add `orca_jobs_by_state` histogram, `orca_drift_events_total` counter, `orca_ssh_errors_total` counter, `orca_txn_apply_total`/`orca_txn_rollback_total` counters, `orca_acl_denials_total` counter, `orca_audit_chain_head` gauge; new `docs/metrics.md` with Prometheus scrape config; security headers middleware on daemon (`X-Content-Type-Options`, `X-Frame-Options`) | Medium | **v0.13 P10** | complete |
|
||||
| REQ-160 | Doc drift round 2: (a) README — update status banner (v0.12+v0.13 complete), latest tag, subcommand table (add `auth`/`nft`/`peer-setup`/`secrets rotate-master`), correct "mTLS by default" claim (SSH-push is canonical, mTLS deprecated), add missing docs to table; (b) `docs/cli.md` — complete rewrite covering all ~40 subcommands; (c) CHANGELOG regen; (d) help text fixes (`job run` HCL→markdown, `job stop` daemon→SSH-push); (e) `docs/webauthn.md` add `auth register`; (f) `docs/namespace.md` add `inherit`/`set-constraint`; (g) `docs/install.md`+`docker.md` update version refs; (h) `docs/security-runbook.md` match P05 reality; (i) fix `verify-reqs` bold-format regex (currently bypasses v0.12); (j) fix ROADMAP/REQUIREMENTS v0.12 status hygiene; (k) `docs/security-scanning.md` gosec.json; (l) `internal/proxmox/bootstrap.go` comments (password→key auth); (m) deprecate `orca status` stub; (n) `make verify-docs` target (cli.md ↔ `orca --help` consistency) | High | **v0.13 P11** | complete |
|
||||
| REQ-161 | `--type linux` SSH-join: implement `NodeKindLinux` path (reserved at `model/node.go:29`); new `internal/linux/bootstrap.go` mirroring Proxmox pattern — orca pubkey deploy → `orca` system user → drift-events dir → no PVE role; key-auth only (R-021); `orca node join --type linux --host <ip> --ssh-user root --ssh-key <path>`; `peer-setup.go` kept as documented fallback | High | **v0.13 P12** | complete |
|
||||
| REQ-162 | UAT plan: `docs/uat.md` — 3-host topology (lead Ubuntu 22.04 + pve01 Proxmox VE 8/9 + worker01 Ubuntu 22.04); step-by-step with exact commands (bootstrap→onboard Proxmox→onboard Ubuntu worker→capacity→namespace→deploy full stack→migrate between hosts→exercise every claim); claim matrix mapping ~35 feature claims to UAT steps; signoff procedure (run `scripts/uat-signoff.sh`, paste output) | Critical | **v0.13 P12** | complete |
|
||||
| REQ-163 | UAT signoff script: `scripts/uat-signoff.sh` — idempotent, `set -euo pipefail`, ~35 named assertions covering all feature claims; read + non-mutating only (doctor, list, --dry-run); exit 0 iff all pass; `scripts/uat-smoke.sh` — pure-CLI subset for CI `validate` (version, acl file mode, doctor modes, no-password grep, metrics shape); tests for both scripts | Critical | **v0.13 P12** | complete |
|
||||
|
||||
### Scope notes (v0.13)
|
||||
|
||||
|
||||
+16
-16
@@ -400,7 +400,7 @@ tags: `v0.10.0`…`v0.10.21`.
|
||||
- External CA / Let's Encrypt / cert transparency
|
||||
- Online-only features (HSTS, OCSP stapling, telemetry)
|
||||
|
||||
## Milestone v0.12: Security Hardening (Zero-Trust Identity) — COMPLETE
|
||||
## Milestone v0.12: Security Hardening (Zero-Trust Identity) — **COMPLETE**
|
||||
|
||||
**Scope**: comprehensive security hardening across the entire attack
|
||||
surface, **including the operating system itself**, plus adoption of a
|
||||
@@ -549,7 +549,7 @@ The v1.0.0 production-ready tag stays deferred for post-v0.12 UAT
|
||||
- External CA / Let's Encrypt / cert transparency
|
||||
- Online-only features (HSTS, OCSP stapling, telemetry)
|
||||
|
||||
## Milestone v0.13: Production Hardening Round 2 + UAT Plan — IN PROGRESS
|
||||
## Milestone v0.13: Production Hardening Round 2 + UAT Plan — **COMPLETE**
|
||||
|
||||
**Scope**: final production hardening round before the v1.0.0
|
||||
production-ready tag. Three deep codebase sweeps (security, reliability,
|
||||
@@ -579,20 +579,20 @@ signoff script that gates the v1.0.0 cut.
|
||||
|
||||
### Phases (14 total: P0 + P01..P12 + P13 final)
|
||||
|
||||
- [ ] Phase P0: Pre-execution (SPECIFY→CLARIFY→RESEARCH→IDEATE→PLAN→GRILL) — tag `v0.12.0`
|
||||
- [ ] Phase P01: Toolchain & dependency vulns (REQ-149) — tag `v0.12.1`
|
||||
- [ ] Phase P02: Input validation & injection hardening (REQ-150) — tag `v0.12.2`
|
||||
- [ ] Phase P03: Scheduler/deployment wiring + jobspec parser (REQ-151, REQ-152) — tag `v0.12.3`
|
||||
- [ ] Phase P04: ACL enforcement + WebAuthn registration auth (REQ-153) — tag `v0.12.4`
|
||||
- [ ] Phase P05: Seal/audit CLI + chain race + key zeroing (REQ-154) — tag `v0.12.5`
|
||||
- [ ] Phase P06: auth init-idp real + auth register (REQ-155) — tag `v0.12.6`
|
||||
- [ ] Phase P07: Concurrency safety (REQ-156) — tag `v0.12.7`
|
||||
- [ ] Phase P08: Transport & SSH safety (REQ-157) — tag `v0.12.8`
|
||||
- [ ] Phase P09: Migration & operational safety (REQ-158) — tag `v0.12.9`
|
||||
- [ ] Phase P10: Observability & metrics (REQ-159) — tag `v0.12.10`
|
||||
- [ ] Phase P11: Doc drift round 2 (REQ-160) — tag `v0.12.11`
|
||||
- [ ] Phase P12: `--type linux` + UAT plan + signoff script (REQ-161, REQ-162, REQ-163) — tag `v0.12.12`
|
||||
- [ ] Phase P13: Final review + ship + audit (milestone release) — tag `v0.12.13` = **v0.13 milestone release**
|
||||
- [x] Phase P0: Pre-execution (SPECIFY→CLARIFY→RESEARCH→IDEATE→PLAN→GRILL) — tag `v0.12.0`
|
||||
- [x] Phase P01: Toolchain & dependency vulns (REQ-149) — tag `v0.12.1`
|
||||
- [x] Phase P02: Input validation & injection hardening (REQ-150) — tag `v0.12.2`
|
||||
- [x] Phase P03: Scheduler/deployment wiring + jobspec parser (REQ-151, REQ-152) — tag `v0.12.3`
|
||||
- [x] Phase P04: ACL enforcement + WebAuthn registration auth (REQ-153) — tag `v0.12.4`
|
||||
- [x] Phase P05: Seal/audit CLI + chain race + key zeroing (REQ-154) — tag `v0.12.5`
|
||||
- [x] Phase P06: auth init-idp real + auth register (REQ-155) — tag `v0.12.6`
|
||||
- [x] Phase P07: Concurrency safety (REQ-156) — tag `v0.12.7`
|
||||
- [x] Phase P08: Transport & SSH safety (REQ-157) — tag `v0.12.8`
|
||||
- [x] Phase P09: Migration & operational safety (REQ-158) — tag `v0.12.9`
|
||||
- [x] Phase P10: Observability & metrics (REQ-159) — tag `v0.12.10`
|
||||
- [x] Phase P11: Doc drift round 2 (REQ-160) — tag `v0.12.11`
|
||||
- [x] Phase P12: `--type linux` + UAT plan + signoff script (REQ-161, REQ-162, REQ-163) — tag `v0.12.12`
|
||||
- [x] Phase P13: Final review + ship + audit (milestone release) — tag `v0.12.13` = **v0.13 milestone release**
|
||||
|
||||
**Milestone tag**: `v0.12.13` (final phase patch = milestone release per
|
||||
feature-milestone rule; no separate `v0.13.0` tag). Per-phase tags:
|
||||
|
||||
+132
-26
@@ -5,31 +5,137 @@ All notable changes to orca are documented in this file.
|
||||
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
|
||||
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
||||
|
||||
- `e1b538575c57158c7a6661d5919b103f6c7932fc` — feat(P06): CoreCI release flow with .coreci.yml and tea integration
|
||||
- `07b8ad2ceaba7ca303dfe91876930d33b76c633e` — ship(P05): health checks merged into milestone
|
||||
- `b06458d31370750417a3b239dac61c6e2fdf5329` — docs(P05): verification - 4 layers pass
|
||||
- `708d9834296271094667700e88e80bfa27db7bdd` — feat(P05): health check daemon with /healthz, /readyz, /v1/* handlers
|
||||
- `30c523c0c7a8e75a2e97b42f1c8a39802febcbdc` — ship(P04): state persistence merged into milestone
|
||||
- `759b1b519d7fadf3d91d3070952d9ad2051a0eba` — docs(P04): verification - 4 layers pass
|
||||
- `b25e074e1d3518f175478184ff8d002ec0d8412c` — feat(P04): audit log + persistence hardening
|
||||
- `bb6b5b3e8342c16601a8503223c7186ecdbb00df` — ship(P03): task exec merged into milestone
|
||||
- `857f7563190e7703f97c607a50d6b0a897d250e9` — docs(P03): verification - 4 layers pass
|
||||
- `f9a98733411cfa8657e82636e0c55671086ebe46` — feat(P03): task execution engine with HCL specs, jobs, tasks, WaitDelay
|
||||
- `78334f1f74f0c185c6d38014c796aac4903b8141` — ship(P02): node mgmt merged into milestone
|
||||
- `c7dbcef9587596786a541a7566479d9fb93fcf0a` — docs(P02): verification - 4 layers pass
|
||||
- `9580f347c68e395dccfbe83b27a857d52bf21075` — feat(P02): node management with SQLite-backed registry
|
||||
- `46e929e4c6539bd604539ba27d5ed0c606e87bb9` — chore(P01): source .env in trigger_coreci.sh for GITEA_TOKEN
|
||||
- `503923bf1ee2c60f8375acc7eb9608d346368e1c` — ship(P01): cli skeleton merged into milestone
|
||||
- `e3f6e1df825d39f73933c9996bd2cc4717ff1061` — docs(P01): verification - 4 layers pass
|
||||
- `aa3cccead503a37dfec75873d06d2d396a2876f2` — feat(P01): CLI skeleton with Cobra, subcommand stubs, pre-push hook
|
||||
- `c2038952c74f7c242ba3be65d2f4269b23685f5a` — docs(P00): create 6 phase plans with wave ordering
|
||||
- `65eb2e601b741b36388598b9f8adddd7bd8dd3a8` — docs(P00): research findings - architecture + personas
|
||||
- `6f34f1794b9f526c06a1dc139d4a74371599502e` — docs(P00): ideation - 30 ideas accepted (3 tiers)
|
||||
- `bc7ce1caf672e87774455a6cd6cc0db986cd09b3` — docs(P00): clarify ambiguities (full autonomy, 10 decisions)
|
||||
- `55aae5347ec09bce9ef7697ea0c9c9ee158bc040` — chore(P00): rename orch-engine to orca, configure gitea + coreci (v0.1)
|
||||
- `0cba1aa5feef9564f8b9a2a97ae735dc859a8a84` — chore(P00): set autonomy level to full
|
||||
- `e2e77e79b9cbfb462044662543845476f843161b` — chore(P00): quick task - populate config.json with backlog reference
|
||||
- `8c086def698bf0af31e8e820b6b7a2783af06f43` — chore(config): populate ciagent config with standard settings
|
||||
- `8774008c3e47e4ca4711f4fef164531006d16216` — docs(init): validate specification
|
||||
## v0.13 milestone (in progress) — tag line v0.12.x
|
||||
|
||||
The v0.13 milestone is **Production Hardening Round 2 + UAT Plan**.
|
||||
Three deep codebase sweeps (security, reliability, feature/doc claims)
|
||||
surfaced ~60 gaps beyond v0.12. v0.13 closes all critical/high/medium
|
||||
findings and delivers the UAT plan + signoff script that gates the
|
||||
v1.0.0 cut.
|
||||
|
||||
**Load-bearing architectural changes**:
|
||||
- **R-022** — `orca job run` deploys to remote nodes via the scheduler →
|
||||
emitter → SSH-push pipeline. The local `exec.CommandContext` path is
|
||||
removed (P03).
|
||||
- **R-023** — Zero-trust enforcement is operationally wired: `acl.Check`
|
||||
is invoked on every daemon handler + sshpush + txn apply path;
|
||||
`acl.json` is 0600; audit `actor` carries OIDC sub/SVID; WebAuthn
|
||||
registration requires auth; `cluster seal`/`unseal` + `doctor audit`/
|
||||
`doctor modes` CLI commands exist (P04, P05).
|
||||
|
||||
### v0.13 phase commits (v0.11.29..HEAD)
|
||||
|
||||
- `ed91d68` — feat(P10): observability expansion — metrics + security headers (REQ-159)
|
||||
- `531b369` — fix(P09): migration + operational safety — job stop, retention, logs cap (REQ-158)
|
||||
- `3a3ea74` — fix(P08): transport + SSH safety — typed errors, IPv6, timeouts, signal (REQ-157)
|
||||
- `0358efe` — fix(P07): concurrency safety — SQLite, flock, cache, atomic writes (REQ-156)
|
||||
- `978334a` — feat(P06): auth init-idp real + auth register + doctor oidc (REQ-155)
|
||||
- `9e83238` — feat(P05): seal/audit CLI + chain race fix + key zeroing (REQ-154)
|
||||
- `5232fcb` — fix(P04): wire ACL enforcement + WebAuthn reg auth + audit actor (REQ-153)
|
||||
- `cf3d98e` — feat(P03): wire scheduler into job run + fix jobspec parser (REQ-151, REQ-152)
|
||||
- `4b70e31` — fix(P02): input validation + injection hardening — 11 vectors (REQ-150)
|
||||
- `b0158c9` — fix(P01): bump go toolchain to 1.25.12 + fix pre-existing test bugs (REQ-149)
|
||||
- `7479cd1` — docs(checkpoint): P0 shipped — v0.12.0 tagged
|
||||
- `1a2dd1a` — docs(P00): incorporate grill binding conditions C-44..C-49
|
||||
- `437d9b2` — docs(P00): grill v0.13 — CONDITIONAL PROCEED (6 binding conditions C-44..C-49)
|
||||
- `82bfab1` — docs(P00): create phase plans — 14 phases, 15 REQs, vertical slices
|
||||
- `a2a651e` — docs(P00): ideation results — 15 accepted (REQ-149..REQ-163), 0 skipped
|
||||
- `3f5e5de` — docs(P00): research findings — threat model round 3 (~60 gaps, F26-F101)
|
||||
- `7a60b35` — docs(P00): clarify v0.13 — 7 decisions resolved (D-248..D-254)
|
||||
- `8071793` — docs(init): validate specification — v0.13 Production Hardening Round 2 + UAT Plan
|
||||
|
||||
### v0.13 phase summary
|
||||
|
||||
- **P0** — Pre-execution: specify → clarify → research → ideate → plan → grill (tag `v0.12.0`)
|
||||
- **P01** — Toolchain & dependency vulns: Go 1.25.12 bump, 24 stdlib vulns closed, govulncheck triage (REQ-149)
|
||||
- **P02** — Input validation & injection hardening: 11 vectors closed (`orca logs --job` RCE, tar-slip, sudoers injection, pprof loopback, txn/nft ID validation, drain allocID, cluster_compat, podman image, nft TrustedProbes, sudoers user/role) (REQ-150)
|
||||
- **P03** — Scheduler/deployment wiring + jobspec parser: `orca job run` wires scheduler → emitter → SSH-push; `schedule:`/`timeout:` parsed by markdown jobspec (REQ-151, REQ-152)
|
||||
- **P04** — ACL enforcement + WebAuthn registration auth: `acl.Check` wired into daemon + sshpush + txn apply; WebAuthn registration requires auth; audit actor carries OIDC sub/SVID (REQ-153)
|
||||
- **P05** — Seal/audit CLI + chain race fix + key zeroing: `orca cluster seal`/`unseal`, `orca doctor audit`, `orca doctor modes` CLI commands; audit hash-chain race fix; master key zeroed on exit (REQ-154)
|
||||
- **P06** — auth init-idp real + auth register + doctor oidc: real Dex deployment, `orca auth register` browser flow, `orca doctor oidc` health check (REQ-155)
|
||||
- **P07** — Concurrency safety: SQLite WAL, flock on known_hosts, cache thread-safety, atomic writes (REQ-156)
|
||||
- **P08** — Transport & SSH safety: typed dial errors, IPv6 support, connect timeouts, signal handling (REQ-157)
|
||||
- **P09** — Migration & operational safety: `orca job stop` via SSH, DB retention check, logs cap (REQ-158)
|
||||
- **P10** — Observability & metrics: metrics endpoint expansion, security headers (REQ-159)
|
||||
- **P11** — Doc drift round 2 (this phase, REQ-160)
|
||||
|
||||
## v0.12 milestone — COMPLETE (tag line v0.11.x)
|
||||
|
||||
The v0.12 milestone is **Security Hardening (Zero-Trust Identity)**.
|
||||
Comprehensive security hardening across the entire attack surface
|
||||
including the OS, plus adoption of a zero-trust identity model. 25
|
||||
threat-model findings (F1..F25) closed. R-021 adopted: no Orca-issued
|
||||
credentials — human identity is exclusively external (OIDC), machine
|
||||
identity is exclusively mTLS/SPIFFE.
|
||||
|
||||
**Milestone release**: `v0.11.28` (29 phases, tags `v0.11.0`..`v0.11.28`).
|
||||
|
||||
### v0.12 phase highlights
|
||||
|
||||
- Command injection fix (REQ-119, F3)
|
||||
- Namespace path traversal fix (REQ-120, F4)
|
||||
- Txn apply path allowlist (REQ-121, F5)
|
||||
- OIDC client + bundled Dex (REQ-144; BYO-IdP override)
|
||||
- WebAuthn connector for Dex / passkeys (REQ-148)
|
||||
- ACL rewrite to OIDC claims + enforcement (REQ-145, REQ-122, F1)
|
||||
- Remove all password/token paths (REQ-146, R-021, C-34)
|
||||
- Master key seal-to-OIDC + Shamir 3-of-5 recovery (REQ-147, C-35)
|
||||
- Daemon auth hardening (REQ-123, REQ-124, F6, F24)
|
||||
- Audit log tamper-evidence (REQ-125, F2)
|
||||
- SVID chain validation (REQ-126, F9)
|
||||
- Backup symlink validation (REQ-127, F7)
|
||||
- step-ca /tmp hardening (REQ-128, F10)
|
||||
- Master key rotation (REQ-129, F12, C-30)
|
||||
- File-mode audit expansion (REQ-130, F13)
|
||||
- aggregate.sh JSON injection + drift-gate fix (REQ-131, F11, F18)
|
||||
- install.sh checksum+GPG verification (REQ-132, F14)
|
||||
- nftables ruleset hardening (REQ-133, F21)
|
||||
- sudoers hardening (REQ-134, F22)
|
||||
- System user consistency (REQ-135, F23)
|
||||
- SQLite file-mode + at-rest encryption (REQ-136, F8, C-31)
|
||||
- Migration safety + identity migration (REQ-137, F19, C-34)
|
||||
- Legacy CA/mTLS/daemon + step-ca password-provisioner deletion (REQ-138, F16)
|
||||
- known_hosts tightening + transport hardening (REQ-139, F15, F25)
|
||||
- Drift event authentication (REQ-140, F18)
|
||||
- Security integration test suite (REQ-141, C-33)
|
||||
- Zero-trust + OIDC + WebAuthn + threat-model docs (REQ-142)
|
||||
- Final review + ship + audit (REQ-143)
|
||||
|
||||
## v0.11 milestone — COMPLETE (tag line v0.10.x)
|
||||
|
||||
The v0.11 milestone is **Production Hardening**. See the git log and
|
||||
ROADMAP for the full phase list.
|
||||
|
||||
## v0.1 milestone — COMPLETE
|
||||
|
||||
Initial CLI skeleton, node management, task execution, state
|
||||
persistence, audit log, health checks, and CoreCI release flow.
|
||||
|
||||
- `e1b5385` — feat(P06): CoreCI release flow with .coreci.yml and tea integration
|
||||
- `07b8ad2` — ship(P05): health checks merged into milestone
|
||||
- `b06458d` — docs(P05): verification - 4 layers pass
|
||||
- `708d983` — feat(P05): health check daemon with /healthz, /readyz, /v1/* handlers
|
||||
- `30c523c` — ship(P04): state persistence merged into milestone
|
||||
- `759b1b5` — docs(P04): verification - 4 layers pass
|
||||
- `b25e074` — feat(P04): audit log + persistence hardening
|
||||
- `bb6b5b3` — ship(P03): task exec merged into milestone
|
||||
- `857f756` — docs(P03): verification - 4 layers pass
|
||||
- `f9a9873` — feat(P03): task execution engine with HCL specs, jobs, tasks, WaitDelay
|
||||
- `78334f1` — ship(P02): node mgmt merged into milestone
|
||||
- `c7dbcef` — docs(P02): verification - 4 layers pass
|
||||
- `9580f34` — feat(P02): node management with SQLite-backed registry
|
||||
- `46e929e` — chore(P01): source .env in trigger_coreci.sh for GITEA_TOKEN
|
||||
- `503923b` — ship(P01): cli skeleton merged into milestone
|
||||
- `e3f6e1d` — docs(P01): verification - 4 layers pass
|
||||
- `aa3ccce` — feat(P01): CLI skeleton with Cobra, subcommand stubs, pre-push hook
|
||||
- `c203895` — docs(P00): create 6 phase plans with wave ordering
|
||||
- `65eb2e6` — docs(P00): research findings - architecture + personas
|
||||
- `6f34f17` — docs(P00): ideation - 30 ideas accepted (3 tiers)
|
||||
- `bc7ce1c` — docs(P00): clarify ambiguities (full autonomy, 10 decisions)
|
||||
- `55aae53` — chore(P00): rename orch-engine to orca, configure gitea + coreci (v0.1)
|
||||
- `0cba1aa` — chore(P00): set autonomy level to full
|
||||
- `e2e77e7` — chore(P00): quick task - populate config.json with backlog reference
|
||||
- `8c086de` — chore(config): populate ciagent config with standard settings
|
||||
- `8774008` — docs(init): validate specification
|
||||
|
||||
Generated by make changelog. Do not edit by hand.
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
.PHONY: build test test-race lint fmt clean run release version changelog help security-scan verify-reqs
|
||||
.PHONY: build test test-race lint fmt clean run release version changelog help security-scan verify-reqs verify-docs
|
||||
|
||||
BINARY := bin/orca
|
||||
GOFLAGS := -trimpath
|
||||
@@ -31,6 +31,7 @@ help:
|
||||
@echo " release Run scripts/release.sh [VERSION] — build, tar, publish"
|
||||
@echo " security-scan Run gosec+govulncheck+gitleaks (P03, REQ-014/027/039)"
|
||||
@echo " verify-reqs Assert ROADMAP COMPLETE ↔ REQUIREMENTS Complete (REQ-060)"
|
||||
@echo " verify-docs Assert docs/cli.md ↔ orca --help consistency (REQ-160)"
|
||||
|
||||
build:
|
||||
@mkdir -p bin
|
||||
@@ -128,3 +129,9 @@ security-scan:
|
||||
# of scope (P04 audit). Exits 0 on consistency, 1 with a diff on drift.
|
||||
verify-reqs:
|
||||
go run ./cmd/verify-reqs .ciagent/ROADMAP.md .ciagent/REQUIREMENTS.md
|
||||
|
||||
# verify-docs asserts that every top-level subcommand in docs/cli.md
|
||||
# exists in `orca --help` output (and vice versa). Catches doc drift
|
||||
# (REQ-160). Requires the binary to be built first (`make build`).
|
||||
verify-docs: build
|
||||
./scripts/verify-docs.sh ./bin/orca docs/cli.md
|
||||
|
||||
@@ -6,8 +6,8 @@ identity.
|
||||
|
||||
## Status
|
||||
|
||||
**v0.11: Production Hardening — IN PROGRESS** | **v1.0: UAT-gated** (cut
|
||||
separately after v0.11 completion per operator decision)
|
||||
**v0.12: Security Hardening (Zero-Trust Identity) — COMPLETE** | **v0.13: Production Hardening Round 2 + UAT Plan — IN PROGRESS** | **v1.0: UAT-gated** (cut
|
||||
separately after v0.13 completion per operator decision)
|
||||
|
||||
See [.ciagent/ROADMAP.md](.ciagent/ROADMAP.md) for the full roadmap.
|
||||
|
||||
@@ -18,8 +18,8 @@ See [.ciagent/ROADMAP.md](.ciagent/ROADMAP.md) for the full roadmap.
|
||||
- **Offline-first** — no cloud dependencies; the cluster is the OS
|
||||
- **CLI-first** — the command line is the primary interface (humans and
|
||||
AI agents)
|
||||
- **Security before features** — mTLS by default; NFRs ship before new
|
||||
functionality
|
||||
- **Security before features** — SSH-push is the canonical transport
|
||||
(mTLS available for daemon mode); NFRs ship before new functionality
|
||||
- **WASM-first** — workloads target OS primitives (systemd units,
|
||||
journald), not a container runtime shim
|
||||
- **Bug fixes before features** — stability is paramount
|
||||
@@ -35,8 +35,8 @@ curl -fsSL https://git.cloudinit.dev/coreci/orca/raw/main/scripts/install.sh | b
|
||||
# System-level install (binary at /usr/local/bin/orca, state at /root/.orca)
|
||||
curl -fsSL https://git.cloudinit.dev/coreci/orca/raw/main/scripts/install.sh | sudo bash -s -- --system
|
||||
|
||||
# Pin a specific version (latest tag: v0.10.19)
|
||||
curl -fsSL https://git.cloudinit.dev/coreci/orca/raw/main/scripts/install.sh | bash -s -- --version v0.10.19
|
||||
# Pin a specific version (latest tag: v0.12.10)
|
||||
curl -fsSL https://git.cloudinit.dev/coreci/orca/raw/main/scripts/install.sh | bash -s -- --version v0.12.10
|
||||
|
||||
# Dry-run: check what would be installed without writing
|
||||
curl -fsSL https://git.cloudinit.dev/coreci/orca/raw/main/scripts/install.sh | bash -s -- --check
|
||||
@@ -65,7 +65,7 @@ config, database, and certificates in the namespace dir:
|
||||
|
||||
```bash
|
||||
curl -fsSL https://git.cloudinit.dev/coreci/orca/raw/main/scripts/install.sh | bash
|
||||
# → "updated orca from v0.8.15 to v0.10.19"
|
||||
# → "updated orca from v0.11.28 to v0.12.10"
|
||||
```
|
||||
|
||||
## Subcommands
|
||||
@@ -73,7 +73,7 @@ curl -fsSL https://git.cloudinit.dev/coreci/orca/raw/main/scripts/install.sh | b
|
||||
| Command | Description |
|
||||
|---------|-------------|
|
||||
| `orca init` | Initialize local orca state with full bootstrap |
|
||||
| `orca status` | Show orca daemon status |
|
||||
| `orca status` | **(deprecated v0.1 stub)** Show orca daemon status — use `orca node list` + `orca metrics /healthz` |
|
||||
| `orca version` | Print version information |
|
||||
| `orca daemon` | **(deprecated)** Run the orca daemon (HTTP API + health checks) |
|
||||
| `orca metrics` | Start metrics endpoint (Prometheus text exposition) |
|
||||
@@ -85,15 +85,18 @@ curl -fsSL https://git.cloudinit.dev/coreci/orca/raw/main/scripts/install.sh | b
|
||||
| `orca job` | Manage orca jobs: `run`, `list`, `stop`, `logs`, `lint`, `verify`, `migrate`, `restart` |
|
||||
| `orca ns` | Manage orca namespaces: `list`, `create`, `delete`, `inspect`, `validate`, `inherit`, `set-constraint` |
|
||||
| `orca cert` | **(deprecated)** Manage orca certificates: `ca-init`, `gen`, `show`, `renew`, `fingerprint` |
|
||||
| `orca doctor` | Run self-checks: `cert`, `network`, `db`, `os`, `proxmox`, `no-orca-on-server` |
|
||||
| `orca doctor` | Run self-checks: `cert`, `network`, `db`, `os`, `proxmox`, `no-orca-on-server`, `nft`, `audit`, `modes`, `oidc`, `db-retention` |
|
||||
| `orca audit` | View orca audit log (`list`) |
|
||||
| `orca cache` | CLI cache management: `show`, `invalidate`, `invalidate-all` |
|
||||
| `orca acl` | ACL management: `grant`, `revoke`, `list`, `check` |
|
||||
| `orca secrets` | Secrets management: `set`, `get`, `list`, `rotate`, `delete` |
|
||||
| `orca secrets` | Secrets management: `set`, `get`, `list`, `rotate`, `delete`, `rotate-master` |
|
||||
| `orca drift` | Drift detection: `show`, `watch`, `acknowledge`, `remediate`, `config` |
|
||||
| `orca txn` | Transaction management: `apply`, `list`, `show`, `rollback` |
|
||||
| `orca nft` | nftables ingress management: `show`, `diff`, `doctor`, `country block`, `rate limit` |
|
||||
| `orca collector` | Collector/aggregator management: `start`, `stop`, `status` |
|
||||
| `orca cluster` | Cluster management: `cutover`, `rotate-lead`, `compat-check` |
|
||||
| `orca cluster` | Cluster management: `cutover`, `rotate-lead`, `compat-check`, `seal`, `unseal` |
|
||||
| `orca auth` | OIDC authentication: `login`, `logout`, `status`, `init-idp`, `register` |
|
||||
| `orca peer-setup` | Create the orca system user + drift-events dir on a peer (REQ-111) |
|
||||
|
||||
See [docs/cli.md](docs/cli.md) for the full CLI reference with all flags
|
||||
and examples.
|
||||
@@ -114,7 +117,7 @@ acknowledged rather than papered over.
|
||||
| Auto-scaling | Cluster autoscaler, HPA/VPA, deep integrations | — |
|
||||
| Daemon footprint | — | No daemon on the critical path; the cluster is the OS |
|
||||
| OS-native | — | Workloads are systemd units + journald; no container runtime shim |
|
||||
| mTLS | — | mTLS by default; no opt-in required |
|
||||
| Transport | — | SSH-push is canonical (no daemon needed); mTLS available for daemon mode |
|
||||
| Offline-first | — | No cloud dependencies; fully air-gapped operation |
|
||||
| WASM-first | — | Workloads target OS primitives, not a container runtime |
|
||||
| Proxmox | — | First-class Proxmox node type (`--type proxmox`) via SSH-push |
|
||||
@@ -129,6 +132,10 @@ acknowledged rather than papered over.
|
||||
| [docs/namespace.md](docs/namespace.md) | Namespace and path layout |
|
||||
| [docs/install.md](docs/install.md) | Installation guide |
|
||||
| [docs/security-scanning.md](docs/security-scanning.md) | Security scanning tools |
|
||||
| [docs/security-runbook.md](docs/security-runbook.md) | Security runbook — seal/unseal, rotation, incident response |
|
||||
| [docs/webauthn.md](docs/webauthn.md) | WebAuthn / passkeys registration and login |
|
||||
| [docs/threat-model.md](docs/threat-model.md) | STRIDE threat model + zero-trust architecture |
|
||||
| [docs/oidc.md](docs/oidc.md) | OIDC configuration — Dex quickstart, BYO IdP |
|
||||
|
||||
## Examples
|
||||
|
||||
@@ -144,6 +151,7 @@ make test # Run tests
|
||||
go vet ./... # Vet all packages
|
||||
make lint # Run gofmt + go vet + shellcheck
|
||||
make verify-reqs # Assert ROADMAP ↔ REQUIREMENTS consistency
|
||||
make verify-docs # Assert docs/cli.md ↔ `orca --help` consistency
|
||||
```
|
||||
|
||||
## Architecture
|
||||
|
||||
+24
-10
@@ -16,17 +16,20 @@ import (
|
||||
// the Phase + Status match at the END of the line, where those two columns
|
||||
// always live. The status token is optionally wrapped in markdown bold
|
||||
// (real rows use `**Complete**`; synthetic/future rows may use bare
|
||||
// `Pending`), and may carry trailing notes (e.g. "**Complete** (P01
|
||||
// shipped v0.2.1)") matched by [^|]* before the closing pipe.
|
||||
var reqRowRe = regexp.MustCompile(`^\|\s*(REQ-\d+)\s*\|.*\|\s*([^|]*?)\s*\|\s*\*{0,2}(Complete|Pending)\*{0,2}[^|]*\|\s*$`)
|
||||
// `pending` or `complete` in any case), and may carry trailing notes
|
||||
// (e.g. "**Complete** (P01 shipped v0.2.1)") matched by [^|]* before
|
||||
// the closing pipe. The (?i) flag makes the match case-insensitive so
|
||||
// lowercase `pending` (used by v0.12/v0.13 REQ rows) is captured;
|
||||
// normalizeStatus canonicalizes the captured value to title case.
|
||||
var reqRowRe = regexp.MustCompile(`(?i)^\|\s*(REQ-\d+)\s*\|.*\|\s*([^|]*?)\s*\|\s*\*{0,2}(Complete|Pending)\*{0,2}[^|]*\|\s*$`)
|
||||
|
||||
// milestoneCompleteRe matches a ROADMAP.md milestone header that is marked
|
||||
// COMPLETE. The bold span is substring-tolerant (GRILL #4): it matches
|
||||
// `**COMPLETE**`, `**COMPLETE (merged to main via v0.3)**`, and any future
|
||||
// variant where the word COMPLETE appears inside the bold span, possibly
|
||||
// preceded or followed by non-asterisk text. The milestone version (v0.X)
|
||||
// is captured.
|
||||
var milestoneCompleteRe = regexp.MustCompile(`^##\s*Milestone\s+(v0\.\d+):.*—\s*\*\*[^*]*\bCOMPLETE\b[^*]*\*\*`)
|
||||
// COMPLETE. The bold markers are optional (GRILL #4 + REQ-160 T11): it
|
||||
// matches `**COMPLETE**`, `**COMPLETE (merged to main via v0.3)**`, and
|
||||
// bare `COMPLETE` (as used by the v0.12 milestone header). The word
|
||||
// COMPLETE may be preceded or followed by non-asterisk text. The milestone
|
||||
// version (v0.X) is captured.
|
||||
var milestoneCompleteRe = regexp.MustCompile(`^##\s*Milestone\s+(v0\.\d+):.*—\s*\*{0,2}[^*]*\bCOMPLETE\b[^*]*\*{0,2}`)
|
||||
|
||||
// phaseRe extracts the milestone version from a REQUIREMENTS Phase cell such
|
||||
// as `v0.7 P1`, `**v0.2 P01**`, `v0.2 P01–P04`, or bare `v0.7`. The cell may
|
||||
@@ -40,6 +43,17 @@ type reqRow struct {
|
||||
status string // "Complete" or "Pending"
|
||||
}
|
||||
|
||||
// normalizeStatus canonicalizes a captured status token to the title-case
|
||||
// form ("Complete" or "Pending") so that case-insensitive matches like
|
||||
// "pending" or "complete" compare correctly against the drift assertions.
|
||||
func normalizeStatus(s string) string {
|
||||
s = strings.TrimSpace(s)
|
||||
if s == "" {
|
||||
return s
|
||||
}
|
||||
return strings.ToUpper(s[:1]) + strings.ToLower(s[1:])
|
||||
}
|
||||
|
||||
// milestoneVersions returns the distinct v0.X milestones referenced in the
|
||||
// phase cell (e.g. "v0.7 P1" → ["v0.7"]; "v0.2 P01 / v0.3 P02" →
|
||||
// ["v0.2","v0.3"]).
|
||||
@@ -174,7 +188,7 @@ func parseRequirements(path string) ([]reqRow, error) {
|
||||
if m == nil {
|
||||
continue
|
||||
}
|
||||
rows = append(rows, reqRow{id: m[1], phase: strings.TrimSpace(m[2]), status: m[3]})
|
||||
rows = append(rows, reqRow{id: m[1], phase: strings.TrimSpace(m[2]), status: normalizeStatus(m[3])})
|
||||
}
|
||||
if err := sc.Err(); err != nil {
|
||||
return nil, err
|
||||
|
||||
+1069
-126
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,52 @@
|
||||
# Orca Metrics Reference
|
||||
|
||||
Orca exposes Prometheus text-exposition metrics at `/metrics` on the
|
||||
metrics endpoint (default `:9100`, configurable via `--addr`).
|
||||
|
||||
## Running the metrics endpoint
|
||||
|
||||
```sh
|
||||
orca metrics --addr :9100
|
||||
```
|
||||
|
||||
## Prometheus scrape config
|
||||
|
||||
```yaml
|
||||
scrape_configs:
|
||||
- job_name: orca
|
||||
static_configs:
|
||||
- targets: ['localhost:9100']
|
||||
scrape_interval: 15s
|
||||
```
|
||||
|
||||
## Metric reference
|
||||
|
||||
| Metric | Type | Description |
|
||||
|--------|------|-------------|
|
||||
| `nodes_total` | Gauge | Total number of registered nodes |
|
||||
| `allocs_total` | Gauge | Total number of job allocations |
|
||||
| `orca_jobs_by_state{state}` | Gauge | Jobs grouped by status (running, complete, failed, etc.) |
|
||||
| `orca_audit_chain_head` | Gauge | Audit chain integrity (1 = chain head verified, 0 = error) |
|
||||
|
||||
## Counter metrics (incremented by CLI operations)
|
||||
|
||||
The following counters are incremented during normal operations and
|
||||
are available when the metrics endpoint polls the DB:
|
||||
|
||||
| Metric | Type | Description |
|
||||
|--------|------|-------------|
|
||||
| `orca_drift_events_total` | Counter | Total drift events detected |
|
||||
| `orca_ssh_errors_total` | Counter | Total SSH connection/exec errors |
|
||||
| `orca_txn_apply_total` | Counter | Total transaction applies |
|
||||
| `orca_txn_rollback_total` | Counter | Total transaction rollbacks |
|
||||
| `orca_acl_denials_total` | Counter | Total ACL denials (enforce mode) |
|
||||
|
||||
## Security headers
|
||||
|
||||
The metrics endpoint sets the following security headers on all responses:
|
||||
- `X-Content-Type-Options: nosniff`
|
||||
- `X-Frame-Options: DENY`
|
||||
|
||||
## Health check
|
||||
|
||||
The endpoint also exposes `/healthz` returning `200 ok` for liveness probes.
|
||||
+62
-13
@@ -8,8 +8,8 @@ directory holds cluster-wide artifacts shared across namespaces.
|
||||
|
||||
> **v0.9 layout (canonical)**: This document describes the v0.9
|
||||
> multi-namespace layout. The v0.8 flat layout (`orca.db`, `ca.crt`,
|
||||
> `server.crt` at the root) is deprecated and will be removed in
|
||||
> v0.11. See [v0.8 flat layout](#deprecated-v08-flat-layout) below.
|
||||
> `server.crt` at the root) is deprecated and removed in v0.12
|
||||
> (REQ-138).
|
||||
|
||||
## Namespace root resolution
|
||||
|
||||
@@ -51,12 +51,16 @@ $ORCA_HOME/
|
||||
├── cluster/ # cluster-wide (NOT a workload namespace)
|
||||
│ ├── ca.crt, ca.key # step-ca root (R-006, D-101)
|
||||
│ ├── master.key # AES-256-GCM root (R-011, mode 0600)
|
||||
│ ├── master.key.sealed # sealed master key (REQ-147, mode 0600)
|
||||
│ ├── config.md # Markdown frontmatter config (R-014)
|
||||
│ ├── known_hosts # SSH known_hosts (D-035)
|
||||
│ ├── orca_ssh_key # orca SSH private key (D-037)
|
||||
│ ├── orca_ssh_key.pub # orca SSH public key
|
||||
│ ├── peers/<host>/ # per-peer directory
|
||||
│ ├── txns/ # cluster transaction log (R-016)
|
||||
│ ├── acl.json # ACL state (mode 0600)
|
||||
│ ├── oidc-client-secret # OIDC client secret (mode 0600, C-36)
|
||||
│ ├── webauthn-credentials.db # WebAuthn public keys (mode 0600)
|
||||
│ └── state/ # cluster state
|
||||
├── _defaults/ # implicit root namespace (always exists)
|
||||
│ ├── ns.md # namespace frontmatter (kind: Namespace)
|
||||
@@ -80,14 +84,16 @@ $ORCA_HOME/
|
||||
exists. Every namespace inherits from `_defaults` and cannot opt out
|
||||
(D-185, D-187).
|
||||
- **`cluster/`** is NOT a workload namespace — it holds cluster-wide
|
||||
artifacts (CA, master key, SSH keys, known_hosts, peers, txns).
|
||||
artifacts (CA, master key, SSH keys, known_hosts, peers, txns, ACL,
|
||||
OIDC secrets, WebAuthn credentials).
|
||||
- **Per-namespace DBs**: each namespace has its own
|
||||
`db/orca.db` (R-002). No namespace column in SQLite.
|
||||
- **Namespace inheritance**: child namespaces inherit env and
|
||||
constraints from parents (via `ns.md` frontmatter `parents:` field).
|
||||
`_defaults` is always appended last in the inheritance chain.
|
||||
- **`orca ns` subcommands**: `list`, `create`, `delete`, `inspect`,
|
||||
`validate` — see [docs/cli.md](cli.md#orca-ns).
|
||||
`validate`, `inherit`, `set-constraint` — see below and
|
||||
[docs/cli.md](cli.md#orca-ns).
|
||||
|
||||
### Path reference (`internal/paths/`)
|
||||
|
||||
@@ -134,6 +140,50 @@ orca ns validate prod
|
||||
orca ns delete staging
|
||||
```
|
||||
|
||||
### `orca ns inherit` — set parent namespace (R-002)
|
||||
|
||||
Set the parent namespace for a namespace. Updates `ns.md` frontmatter
|
||||
(`parents` field) and validates the new chain has no cycles. The
|
||||
implicit root `_defaults` is always appended last (D-185).
|
||||
|
||||
```bash
|
||||
orca ns inherit <name> --parent <parent-namespace>
|
||||
```
|
||||
|
||||
**Example**:
|
||||
```bash
|
||||
# Make staging inherit from prod (chain: staging -> prod -> _defaults)
|
||||
orca ns inherit staging --parent prod
|
||||
```
|
||||
|
||||
The child cannot inherit from itself transitively — the resolver
|
||||
validates the chain before writing. If a cycle is detected, the
|
||||
command exits 1 with an error.
|
||||
|
||||
### `orca ns set-constraint` — set a constraint (R-002)
|
||||
|
||||
Set a constraint on a namespace. Constraints are `key=value` strings
|
||||
(e.g., `max-allocs=10`) stored in `ns.md` frontmatter and unioned
|
||||
across the inheritance chain by the resolver.
|
||||
|
||||
```bash
|
||||
orca ns set-constraint <name> <key>=<value>
|
||||
```
|
||||
|
||||
**Example**:
|
||||
```bash
|
||||
# Limit prod to 10 concurrent allocations
|
||||
orca ns set-constraint prod max-allocs=10
|
||||
|
||||
# Set a required node affinity
|
||||
orca ns set-constraint prod require-label=ssd
|
||||
```
|
||||
|
||||
Constraints are unioned (not overridden) across the inheritance chain:
|
||||
if `_defaults` sets `max-allocs=50` and `prod` sets `max-allocs=10`,
|
||||
the effective constraint is the most restrictive one (CEL evaluation
|
||||
determines precedence per constraint key).
|
||||
|
||||
See [docs/cli.md](cli.md#orca-ns) for the full `orca ns` reference.
|
||||
|
||||
## `ORCA_DB` override
|
||||
@@ -148,11 +198,10 @@ orca init # uses /tmp/test.db for the DB, ~/.orca/ for everything else
|
||||
|
||||
## Deprecated: v0.8 flat layout
|
||||
|
||||
> **Deprecated in v0.9**: The v0.8 flat layout (`orca.db`, `ca.crt`,
|
||||
> `ca.key`, `server.crt`, `server.key` at the namespace root) is
|
||||
> superseded by the v0.9 multi-namespace layout (R-002). The v0.8
|
||||
> layout is supported during the dual-write window via
|
||||
> `internal/certpaths` (a thin shim) and will be removed in v0.11.
|
||||
> **Removed in v0.12** (REQ-138): The v0.8 flat layout (`orca.db`,
|
||||
> `ca.crt`, `ca.key`, `server.crt`, `server.key` at the namespace root)
|
||||
> is superseded by the v0.9 multi-namespace layout (R-002) and the
|
||||
> dual-write window is closed.
|
||||
|
||||
The v0.8 flat layout stored all state at the namespace root:
|
||||
|
||||
@@ -166,12 +215,12 @@ The v0.8 flat layout stored all state at the namespace root:
|
||||
|
||||
The v0.9 re-architecture moved these to `cluster/` (CA, SSH keys) and
|
||||
per-namespace `db/` (SQLite) to support multi-tenancy (R-002). The
|
||||
`orca doctor --legacy-paths` command (v0.11-P14c) will detect v0.8
|
||||
residue and recommend migration.
|
||||
`internal/certpaths` shim that supported the dual-write window is
|
||||
removed in v0.12.
|
||||
|
||||
## See also
|
||||
|
||||
- [Install Guide](install.md) — 1-liner install with `install.sh`.
|
||||
- [Docker Guide](docker.md) — running orca in a container.
|
||||
- [CLI Reference](cli.md) — `orca ns` subcommands.
|
||||
- [Jobspec Reference](jobspec.md) — markdown frontmatter schema.
|
||||
- [CLI Reference](cli.md#orca-ns) — `orca ns` subcommands.
|
||||
- [Jobspec Reference](jobspec.md) — markdown frontmatter schema.
|
||||
|
||||
+138
-21
@@ -1,31 +1,148 @@
|
||||
# Security Runbook (v0.12)
|
||||
# Security Runbook (v0.13)
|
||||
|
||||
## Master Key Seal/Unseal
|
||||
This runbook documents the operational security procedures for orca's
|
||||
zero-trust identity model (R-021): human identity is exclusively
|
||||
external (OIDC), machine identity is exclusively mTLS/SPIFFE, and no
|
||||
passwords / Orca-issued tokens / CA-key passphrases exist anywhere in
|
||||
the system. The v0.12 milestone shipped these capabilities; the v0.13
|
||||
milestone wired them operationally (R-023).
|
||||
|
||||
- `orca cluster seal`: encrypts master key with OIDC-derived key;
|
||||
prints 5 Shamir shards for offline recovery.
|
||||
- `orca cluster unseal`: operator authenticates via OIDC; master key
|
||||
unwrapped into memory; zeroed on shutdown.
|
||||
- `orca cluster unseal --recovery`: if IdP lost, present 3 of 5 shards.
|
||||
## Master Key Seal/Unseal (REQ-147, P05)
|
||||
|
||||
## Master Key Rotation
|
||||
The cluster master key (`ClusterDir()/master.key`, mode 0600) encrypts
|
||||
all namespace `.env.secrets` via per-namespace HKDF-SHA256 sub-keys
|
||||
(AES-256-GCM). The master key can be **sealed** (encrypted at rest) and
|
||||
**unsealed** (unwrapped into memory for use).
|
||||
|
||||
`orca secrets rotate-master [--dry-run]`: generates new master key,
|
||||
re-encrypts all namespace secrets, re-seals. Atomic + automatic rollback.
|
||||
### Seal
|
||||
|
||||
```bash
|
||||
orca cluster seal
|
||||
```
|
||||
|
||||
Encrypts the raw master key with a key derived from either:
|
||||
- the OIDC ID token subject (if `orca auth login` has been run), or
|
||||
- the cluster CA fingerprint (mTLS-only offline path, D-241).
|
||||
|
||||
The sealed blob is written to `ClusterDir()/master.key.sealed` (0600).
|
||||
**Five Shamir shards (3-of-5 recovery)** are printed to stdout — store
|
||||
them offline. The raw master key is then deleted from disk so the
|
||||
cluster is sealed at rest.
|
||||
|
||||
### Unseal
|
||||
|
||||
```bash
|
||||
orca cluster unseal
|
||||
```
|
||||
|
||||
Reads the sealed blob and unwraps the master key using the OIDC ID
|
||||
token subject or the cluster CA fingerprint. The unwrapped key is
|
||||
written back to `ClusterDir()/master.key` (0600) and zeroed from
|
||||
memory on process exit.
|
||||
|
||||
### Recovery (IdP lost)
|
||||
|
||||
```bash
|
||||
orca cluster unseal --recovery
|
||||
```
|
||||
|
||||
If the IdP is permanently lost, the operator is prompted for 3 of the
|
||||
5 Shamir shards printed at seal time. With quorum, the master key is
|
||||
reconstructed and written back to disk. If quorum is unavailable, the
|
||||
cluster is unrecoverable by design (C-35: no backdoor).
|
||||
|
||||
## Master Key Rotation (REQ-129, C-30)
|
||||
|
||||
```bash
|
||||
orca secrets rotate-master [--dry-run]
|
||||
```
|
||||
|
||||
Generates a new master key, re-encrypts every namespace's
|
||||
`.env.secrets` under the new key, and re-seals the master key to OIDC.
|
||||
With `--dry-run`, reports affected namespaces without writing.
|
||||
|
||||
- **Atomic per-namespace**: each namespace is re-encrypted independently.
|
||||
- **Automatic rollback**: on any namespace failure, the old sealed key
|
||||
is restored (C-30).
|
||||
- **No passphrase** (R-021): the master key is sealed to OIDC, not to a
|
||||
human-typed passphrase.
|
||||
|
||||
## File-Mode Audit (REQ-033, REQ-130, F13)
|
||||
|
||||
```bash
|
||||
orca doctor modes
|
||||
```
|
||||
|
||||
Verifies file modes on security-sensitive files across `ORCA_HOME`:
|
||||
- private keys / secrets: `0600`
|
||||
- certs / public keys: `0644`
|
||||
|
||||
Exits 0 if all files have correct modes; exits 1 if any violation is
|
||||
found. Missing files are not counted as violations.
|
||||
|
||||
Checks: SSH key, master key (sealed blob), server cert/key,
|
||||
known_hosts, `acl.json`, OIDC client secret.
|
||||
|
||||
## Audit Log Tamper-Evidence (REQ-125, F2)
|
||||
|
||||
```bash
|
||||
orca doctor audit
|
||||
```
|
||||
|
||||
Verifies the audit log hash chain. Opens the orca SQLite DB, recomputes
|
||||
the hash chain from the first audit entry, and reports the chain head
|
||||
hash. If any entry's `entry_hash` or `prev_hash` link does not match the
|
||||
recomputed value, the chain has been tampered with and the command
|
||||
exits non-zero.
|
||||
|
||||
The audit log is append-only (SQLite trigger blocks
|
||||
UPDATE/DELETE). Each entry's `actor` field carries the OIDC `sub` or
|
||||
SPIFFE SVID. Run this after any suspected intrusion or as part of a
|
||||
regular audit cadence.
|
||||
|
||||
## Sudoers Audit (REQ-134, F22)
|
||||
|
||||
```bash
|
||||
orca doctor proxmox
|
||||
```
|
||||
|
||||
Audits the `/etc/sudoers.d/orca` file against the expected allowlist:
|
||||
- `pct` + `qm` with NOEXEC
|
||||
- `apt-get` / `dpkg` excluded (or NOEXEC'd)
|
||||
- `pvesh` EXCLUDED (AD-020: pvesh can bypass NOEXEC via the API execute
|
||||
endpoint)
|
||||
|
||||
## nft Audit (REQ-133, F21)
|
||||
|
||||
```bash
|
||||
orca doctor nft
|
||||
```
|
||||
|
||||
Audits the live nftables ingress ruleset against the on-disk
|
||||
`/etc/nftables.d/orca.nft` hash (recorded at the latest applied txn).
|
||||
Reports drift if the live ruleset does not match. Also verifies:
|
||||
- table exists
|
||||
- DNAT `:443 → 127.0.0.1:8443` and `:80 → 127.0.0.1:8080` present
|
||||
- rate-limit meter present
|
||||
- `/etc/nftables.d/orca.nft` parses
|
||||
|
||||
## Incident Response
|
||||
|
||||
1. Revoke the compromised identity (OIDC user/group or SPIFFE SVID).
|
||||
2. Rotate the master key (`orca secrets rotate-master`).
|
||||
3. Review the audit log (`orca doctor audit` verifies the hash chain).
|
||||
4. If the master key is compromised, all historical secrets are
|
||||
compromised (no forward secrecy).
|
||||
1. **Revoke the compromised identity** (OIDC user/group or SPIFFE SVID).
|
||||
2. **Rotate the master key** (`orca secrets rotate-master`).
|
||||
3. **Review the audit log** (`orca doctor audit` verifies the hash
|
||||
chain; `orca audit list` shows entries).
|
||||
4. **Check file modes** (`orca doctor modes` detects permission drift).
|
||||
5. If the master key is compromised, **all historical secrets are
|
||||
compromised** (no forward secrecy — documented residual risk).
|
||||
6. **Re-seal** the master key after rotation (`orca cluster seal`).
|
||||
|
||||
## Sudoers Audit
|
||||
## OIDC Provider Health (P06)
|
||||
|
||||
`orca doctor proxmox` audits the `/etc/sudoers.d/orca` file against the
|
||||
expected allowlist (pct + qm with NOEXEC; apt-get/dpkg excluded).
|
||||
```bash
|
||||
orca doctor oidc
|
||||
```
|
||||
|
||||
## nft Audit
|
||||
|
||||
`orca doctor nft` audits the live nftables ruleset against the emitted one.
|
||||
Checks the bundled Dex OIDC provider health. Verifies the Dex systemd
|
||||
unit is running and the `/.well-known/openid-configuration` endpoint
|
||||
responds. Run after `orca auth init-idp` or after a Dex config change.
|
||||
|
||||
+319
@@ -0,0 +1,319 @@
|
||||
# Orca User Acceptance Testing (UAT) Plan
|
||||
|
||||
**Version**: v0.13 (production hardening round 2)
|
||||
**Gate**: v1.0.0 production-ready tag is deferred until this UAT passes
|
||||
**Signoff**: run `scripts/uat-signoff.sh` on the lead node and paste the output back
|
||||
|
||||
## Prerequisites
|
||||
|
||||
### Hardware
|
||||
|
||||
| Role | OS | Requirements |
|
||||
|------|-----|-------------|
|
||||
| **lead** | Ubuntu 22.04 LTS | Operator laptop or VM; SSH key; `orca` binary (built from v0.13 tag) |
|
||||
| **pve01** | Proxmox VE 8/9 | Bare-metal or nested; SSH root access; orca SSH key pre-staged |
|
||||
| **worker01** | Ubuntu 22.04 LTS | VM or bare-metal; SSH root access; orca SSH key pre-staged |
|
||||
|
||||
### Alternative topology (3x Ubuntu, no Proxmox)
|
||||
|
||||
If a Proxmox host is unavailable, run the UAT with 3x Ubuntu hosts.
|
||||
Use `--type linux` for all remote nodes. Proxmox-specific claims
|
||||
(`doctor proxmox`, PVE role, sudoers) are **skipped** in this path.
|
||||
The signoff script reports exercised vs. skipped claims.
|
||||
|
||||
### Pre-staging
|
||||
|
||||
1. Build orca from the v0.13 tag:
|
||||
```sh
|
||||
git clone https://git.cloudinit.dev/coreci/orca.git
|
||||
cd orca && git checkout v0.12.13
|
||||
make build
|
||||
# binary is at bin/orca
|
||||
```
|
||||
|
||||
2. Generate the orca SSH keypair on the lead:
|
||||
```sh
|
||||
ssh-keygen -t ed25519 -f ~/.ssh/orca_ed25519 -N ""
|
||||
```
|
||||
|
||||
3. Pre-stage the orca public key on pve01 and worker01:
|
||||
```sh
|
||||
ssh-copy-id -i ~/.ssh/orca_ed25519.pub root@pve01
|
||||
ssh-copy-id -i ~/.ssh/orca_ed25519.pub root@worker01
|
||||
```
|
||||
|
||||
4. Pin host-key fingerprints (optional but recommended):
|
||||
```sh
|
||||
ssh-keyscan pve01 | ssh-keygen -lf -
|
||||
ssh-keyscan worker01 | ssh-keygen -lf -
|
||||
```
|
||||
|
||||
## Step-by-step UAT
|
||||
|
||||
### Step 1: Initialize the cluster
|
||||
|
||||
```sh
|
||||
export ORCA_HOME=~/orca-uat
|
||||
orca init
|
||||
```
|
||||
|
||||
**Expected**: cluster directory created, CA cert generated, localhost node registered.
|
||||
|
||||
### Step 2: Onboard the Proxmox host
|
||||
|
||||
```sh
|
||||
orca node join --type proxmox \
|
||||
--host pve01 \
|
||||
--ssh-user root \
|
||||
--ssh-key ~/.ssh/orca_ed25519 \
|
||||
--host-key-fingerprint SHA256:<fingerprint>
|
||||
```
|
||||
|
||||
**Expected**: SSH bootstrap succeeds, orca user created, PVE role assigned, node registered as `ready` with `kind=proxmox`.
|
||||
|
||||
### Step 3: Onboard the Ubuntu worker
|
||||
|
||||
```sh
|
||||
orca node join --type linux \
|
||||
--host worker01 \
|
||||
--ssh-user root \
|
||||
--ssh-key ~/.ssh/orca_ed25519 \
|
||||
--host-key-fingerprint SHA256:<fingerprint>
|
||||
```
|
||||
|
||||
**Expected**: SSH bootstrap succeeds, orca user created, drift-events dir created, node registered as `ready` with `kind=linux`.
|
||||
|
||||
### Step 4: Verify nodes
|
||||
|
||||
```sh
|
||||
orca node list
|
||||
orca node list --json
|
||||
```
|
||||
|
||||
**Expected**: 3 nodes listed (localhost + pve01 + worker01), all `ready`.
|
||||
|
||||
### Step 5: Set capacity on remote nodes
|
||||
|
||||
```sh
|
||||
orca node capacity set --node pve01 --cpu 4 --memory 8192 --disk 100000
|
||||
orca node capacity set --node worker01 --cpu 2 --memory 4096 --disk 50000
|
||||
orca node capacity list
|
||||
```
|
||||
|
||||
**Expected**: capacity shown for both remote nodes.
|
||||
|
||||
### Step 6: Create a namespace
|
||||
|
||||
```sh
|
||||
orca ns create prod
|
||||
orca ns list
|
||||
```
|
||||
|
||||
**Expected**: `prod` namespace listed.
|
||||
|
||||
### Step 7: Deploy the full stack
|
||||
|
||||
Deploy each service from `examples/full-stack/`:
|
||||
|
||||
```sh
|
||||
orca job run examples/full-stack/web-app.md --target pve01
|
||||
orca job run examples/full-stack/api.md --target pve01
|
||||
orca job run examples/full-stack/worker.md --target worker01
|
||||
orca job run examples/full-stack/postgres.md --target pve01
|
||||
orca job run examples/full-stack/log-shipper.md --target worker01
|
||||
```
|
||||
|
||||
**Expected**: each job is scheduled on the target, systemd unit deployed via SSH-push, job status `running` or `complete`.
|
||||
|
||||
### Step 8: Verify deployment
|
||||
|
||||
```sh
|
||||
orca job list
|
||||
orca job list --json
|
||||
```
|
||||
|
||||
**Expected**: all 5 jobs listed, with correct target nodes.
|
||||
|
||||
On each remote node:
|
||||
```sh
|
||||
ssh root@pve01 systemctl status 'orca-alloc-*'
|
||||
ssh root@worker01 systemctl status 'orca-alloc-*'
|
||||
```
|
||||
|
||||
### Step 9: Verify Traefik routes
|
||||
|
||||
```sh
|
||||
ssh root@pve01 ls /etc/traefik/dynamic/
|
||||
ssh root@worker01 ls /etc/traefik/dynamic/
|
||||
```
|
||||
|
||||
**Expected**: `traefik-dynamic-*.yaml` files present on nodes where jobs were deployed.
|
||||
|
||||
### Step 10: Migrate between hosts
|
||||
|
||||
Migrate `web-app` from pve01 to worker01:
|
||||
|
||||
```sh
|
||||
orca job migrate web-app --to worker01
|
||||
```
|
||||
|
||||
**Expected**: job drained on pve01, rescheduled on worker01, new systemd unit deployed.
|
||||
|
||||
Verify:
|
||||
```sh
|
||||
orca job list
|
||||
ssh root@worker01 systemctl status 'orca-alloc-*web-app*'
|
||||
ssh root@pve01 systemctl status 'orca-alloc-*web-app*' # should be stopped
|
||||
```
|
||||
|
||||
### Step 11: Aggregate logs
|
||||
|
||||
```sh
|
||||
orca logs --all-nodes --job web-app --since 5m
|
||||
```
|
||||
|
||||
**Expected**: log entries from multiple nodes.
|
||||
|
||||
### Step 12: ACL enforcement
|
||||
|
||||
```sh
|
||||
orca acl grant operator-1 --namespace prod --permissions read,write
|
||||
orca acl check operator-1 --namespace prod --permission read
|
||||
orca acl check operator-1 --namespace prod --permission admin
|
||||
```
|
||||
|
||||
**Expected**: read+write allowed, admin denied (not granted).
|
||||
|
||||
### Step 13: Seal/unseal
|
||||
|
||||
```sh
|
||||
orca cluster seal --rp-id orca.local
|
||||
orca cluster unseal
|
||||
orca secrets set prod TEST_KEY --value "test-value"
|
||||
orca secrets get prod TEST_KEY
|
||||
```
|
||||
|
||||
**Expected**: seal succeeds, unseal succeeds, secrets readable post-unseal.
|
||||
|
||||
### Step 14: Audit chain
|
||||
|
||||
```sh
|
||||
orca doctor audit
|
||||
```
|
||||
|
||||
**Expected**: chain head reported, no tamper detected.
|
||||
|
||||
### Step 15: Doctor modes
|
||||
|
||||
```sh
|
||||
orca doctor modes
|
||||
```
|
||||
|
||||
**Expected**: all file modes correct, exit 0.
|
||||
|
||||
### Step 16: OIDC health
|
||||
|
||||
```sh
|
||||
orca doctor oidc
|
||||
```
|
||||
|
||||
**Expected**: Dex unit active, issuer reachable (or WARN if Dex not installed).
|
||||
|
||||
### Step 17: Backup and restore
|
||||
|
||||
```sh
|
||||
orca backup --out /tmp/uat-backup.tar.gz
|
||||
orca restore --in /tmp/uat-backup.tar.gz --dry-run
|
||||
```
|
||||
|
||||
**Expected**: backup succeeds, restore dry-run succeeds.
|
||||
|
||||
### Step 18: Drift detection
|
||||
|
||||
```sh
|
||||
orca drift show
|
||||
```
|
||||
|
||||
**Expected**: no error (empty drift is fine).
|
||||
|
||||
### Step 19: Transaction idempotency
|
||||
|
||||
```sh
|
||||
orca txn apply <some-txn-dir>
|
||||
orca txn apply <some-txn-dir> # re-run
|
||||
```
|
||||
|
||||
**Expected**: second apply is idempotent (exit 5 or "already applied").
|
||||
|
||||
### Step 20: Metrics
|
||||
|
||||
```sh
|
||||
orca metrics --addr :9100 &
|
||||
sleep 3
|
||||
curl -s http://localhost:9100/metrics | grep orca_
|
||||
```
|
||||
|
||||
**Expected**: expanded metric set present (`orca_jobs_running`, `orca_audit_chain_head`, etc.).
|
||||
|
||||
### Step 21: Compat check
|
||||
|
||||
```sh
|
||||
orca cluster compat-check
|
||||
```
|
||||
|
||||
**Expected**: exit 0, all nodes compatible.
|
||||
|
||||
### Step 22: Run the signoff script
|
||||
|
||||
```sh
|
||||
scripts/uat-signoff.sh
|
||||
```
|
||||
|
||||
**Expected**: `UAT SIGNOFF: N/35 assertions passed`, exit 0 iff N==35.
|
||||
|
||||
## Claim Matrix
|
||||
|
||||
| # | Claim | UAT Step | Signoff Assertion |
|
||||
|---|-------|----------|-------------------|
|
||||
| 1 | Cluster initializes from scratch | Step 1 | `assert_orca_version` |
|
||||
| 2 | Proxmox host onboards via SSH | Step 2 | `assert_proxmox_onboarded` |
|
||||
| 3 | Ubuntu worker onboards via `--type linux` | Step 3 | `assert_linux_worker_onboarded` |
|
||||
| 4 | Node list shows all nodes | Step 4 | `assert_cluster_initialized` |
|
||||
| 5 | Capacity is set on remote nodes | Step 5 | `assert_capacity_set` |
|
||||
| 6 | Namespace created | Step 6 | `assert_namespace_created` |
|
||||
| 7 | Full stack deploys to remote nodes | Step 7 | `assert_full_stack_running` |
|
||||
| 8 | Scheduler deploys to remote (not local) | Step 7 | `assert_job_deploys_to_remote` |
|
||||
| 9 | Traefik routes present | Step 9 | `assert_traefik_routes` |
|
||||
| 10 | Job migrates between hosts | Step 10 | `assert_migrate_worked` |
|
||||
| 11 | Logs aggregate from multiple nodes | Step 11 | `assert_logs_aggregate` |
|
||||
| 12 | ACL grant/check works | Step 12 | `assert_acl_enforced` |
|
||||
| 13 | ACL deny-by-default | Step 12 | `assert_acl_deny_default` |
|
||||
| 14 | acl.json mode 0600 | Step 12 | `assert_acl_file_mode` |
|
||||
| 15 | Seal/unseal round-trip | Step 13 | `assert_seal_unseal_roundtrip` |
|
||||
| 16 | Audit chain intact | Step 14 | `assert_audit_chain_intact` |
|
||||
| 17 | Doctor modes passes | Step 15 | `assert_doctor_modes` |
|
||||
| 18 | OIDC health check | Step 16 | `assert_oidc_health` |
|
||||
| 19 | Backup works | Step 17 | `assert_backup_restore_dryrun` |
|
||||
| 20 | Drift visible | Step 18 | `assert_drift_visible` |
|
||||
| 21 | Txn idempotent | Step 19 | `assert_txn_idempotent` |
|
||||
| 22 | Metrics expanded | Step 20 | `assert_metrics_expanded` |
|
||||
| 23 | Compat check passes | Step 21 | `assert_compat_check_passes` |
|
||||
| 24 | No `--password` in docs/examples | — | `assert_no_password_in_docs` |
|
||||
| 25 | Go toolchain current | — | `assert_go_toolchain_current` |
|
||||
| 26 | cli.md matches `orca --help` | — | `assert_cli_md_complete` |
|
||||
| 27 | pprof not on all interfaces | — | `assert_no_pprof_on_all_interfaces` |
|
||||
| 28 | WebAuthn registration requires auth | — | `assert_webauthn_reg_requires_auth` |
|
||||
| 29 | Audit chain survives concurrency | — | `assert_audit_chain_concurrent` |
|
||||
| 30 | Concurrent secrets no data loss | — | `assert_concurrent_secrets_no_loss` |
|
||||
| 31 | Cache invalidated after write | — | `assert_cache_invalidated_after_write` |
|
||||
| 32 | SQLite no lock under concurrency | — | `assert_sqlite_no_lock` |
|
||||
| 33 | No injection in logs --job | — | `assert_no_injection_in_logs` |
|
||||
| 34 | `--type linux` exists as subcommand | Step 3 | `assert_type_linux_available` |
|
||||
| 35 | `orca status` deprecated | — | `assert_status_deprecated` |
|
||||
|
||||
## Signoff procedure
|
||||
|
||||
1. Run all steps above on the 3-host cluster
|
||||
2. Run `scripts/uat-signoff.sh` on the lead
|
||||
3. Paste the output back to the CI agent
|
||||
4. The CI agent verifies `35/35 PASS` and cuts `v1.0.0`
|
||||
+84
-14
@@ -1,27 +1,97 @@
|
||||
# WebAuthn / Passkeys (v0.12)
|
||||
# WebAuthn / Passkeys (v0.13)
|
||||
|
||||
## Overview
|
||||
|
||||
The bundled Dex uses a custom WebAuthn connector for password-free
|
||||
authentication. Passkeys are public-key credentials — the private key
|
||||
never leaves the authenticator (TPM/security key/phone Secure Enclave).
|
||||
The bundled Dex uses a custom WebAuthn connector (`orca-webauthn-connector`,
|
||||
REQ-148) for password-free authentication. Passkeys are public-key
|
||||
credentials — the private key never leaves the authenticator (TPM /
|
||||
security key / phone Secure Enclave). This directly satisfies R-021
|
||||
(no Orca-issued credentials): the authenticator proves possession of
|
||||
the private key without ever exposing it.
|
||||
|
||||
The WebAuthn connector ships as part of the v0.12 milestone (P05) and
|
||||
is operationally wired in v0.13 (P04: registration requires auth; P06:
|
||||
real Dex deployment).
|
||||
|
||||
## Registration
|
||||
|
||||
`orca auth register` opens the browser to the Dex WebAuthn endpoint.
|
||||
After the ceremony (biometric/security key), Dex maps the credential
|
||||
ID to an OIDC `sub`. Credentials stored at
|
||||
`ClusterDir()/webauthn-credentials.db` (0600, public keys only).
|
||||
```bash
|
||||
orca auth register [--no-browser]
|
||||
```
|
||||
|
||||
Opens the browser to the Dex WebAuthn registration page at
|
||||
`https://<cluster>/orca/webauthn/register`. The operator authenticates
|
||||
via an existing session or admin bootstrap token, then performs the
|
||||
WebAuthn ceremony (biometric or security key). After the ceremony,
|
||||
Dex maps the credential ID to an OIDC `sub`.
|
||||
|
||||
- **`--no-browser`**: print the registration URL instead of opening a
|
||||
browser (useful for headless operators or remote SSH sessions — copy
|
||||
the URL into a local browser).
|
||||
|
||||
Credentials are stored at `ClusterDir()/webauthn-credentials.db`
|
||||
(mode 0600, public keys only — private keys never leave the
|
||||
authenticator and are never stored by orca).
|
||||
|
||||
**Example**:
|
||||
```bash
|
||||
# Interactive (opens browser)
|
||||
orca auth register
|
||||
|
||||
# Headless / remote SSH (print URL)
|
||||
orca auth register --no-browser
|
||||
# → https://orca.local/orca/webauthn/register
|
||||
```
|
||||
|
||||
## RP ID
|
||||
|
||||
The relying-party ID is the cluster's Traefik-served domain
|
||||
(`--rp-id` on `orca auth init-idp`). HTTPS secure context is provided
|
||||
by Traefik (step-ca cert, R-017).
|
||||
The relying-party ID is the cluster's Traefik-served domain, set via
|
||||
`--rp-id` on `orca auth init-idp` (C-38). The RP ID **must** match the
|
||||
cluster's Traefik domain — WebAuthn enforces that the RP ID is a
|
||||
registrable domain suffix of the current origin.
|
||||
|
||||
HTTPS secure context is provided by Traefik (step-ca cert, R-017).
|
||||
WebAuthn requires a secure context (HTTPS or localhost); the step-ca
|
||||
cert behind Traefik satisfies this.
|
||||
|
||||
## Bootstrap Sequence
|
||||
|
||||
1. `orca init` bootstraps the cluster CA (step-ca, mTLS-only).
|
||||
2. `orca auth init-idp` deploys Dex behind Traefik (step-ca cert).
|
||||
3. First operator registers a passkey via the mTLS-authenticated session.
|
||||
4. Subsequent operators use WebAuthn.
|
||||
2. `orca auth init-idp --rp-id <cluster-domain>` deploys Dex behind
|
||||
Traefik (step-ca cert) with the WebAuthn connector configured.
|
||||
3. First operator authenticates via an existing session or admin
|
||||
bootstrap token, then registers a passkey:
|
||||
|
||||
```bash
|
||||
orca auth register
|
||||
```
|
||||
|
||||
4. Subsequent operators use WebAuthn login (`orca auth login` opens
|
||||
the browser to the Dex login page; the WebAuthn ceremony is one of
|
||||
the available upstreams).
|
||||
|
||||
## Health Check
|
||||
|
||||
```bash
|
||||
orca doctor oidc
|
||||
```
|
||||
|
||||
Verifies the bundled Dex OIDC provider is running and the
|
||||
`/.well-known/openid-configuration` endpoint responds. Run after
|
||||
`orca auth init-idp` or after a Dex config change.
|
||||
|
||||
## Security properties
|
||||
|
||||
- **No passwords**: WebAuthn is password-free. No password is ever
|
||||
sent to or stored by orca (R-021).
|
||||
- **Phishing-resistant**: the WebAuthn protocol cryptographically binds
|
||||
the ceremony to the RP ID, defeating credential phishing.
|
||||
- **Private key never leaves the authenticator**: orca stores only
|
||||
public keys.
|
||||
- **Secure context required**: HTTPS via step-ca / Traefik (C-38).
|
||||
|
||||
## See also
|
||||
|
||||
- [docs/oidc.md](oidc.md) — OIDC configuration (Dex quickstart, BYO IdP)
|
||||
- [docs/security-runbook.md](security-runbook.md) — security runbook
|
||||
- [docs/cli.md](cli.md#orca-auth) — `orca auth` CLI reference
|
||||
|
||||
@@ -62,7 +62,7 @@ a localhost node.
|
||||
orca node join --type proxmox --host 192.168.1.100 --ssh-user root
|
||||
|
||||
# Join a second node
|
||||
ORCA_PROXMOX_PASSWORD=secret orca node join --type proxmox --host 192.168.1.101
|
||||
orca node join --type proxmox --host 192.168.1.101 --ssh-key ~/.ssh/orca_ed25519
|
||||
```
|
||||
|
||||
### Step 3: Declare node capacity
|
||||
|
||||
Vendored
+13
-1
@@ -58,14 +58,26 @@ func Open(path string) (*Cache, error) {
|
||||
if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {
|
||||
return nil, fmt.Errorf("create cache db dir: %w", err)
|
||||
}
|
||||
db, err := sql.Open("sqlite", path+"?_pragma=journal_mode(WAL)")
|
||||
// REQ-156 / P07 T1: busy_timeout(5000) so concurrent cache opens
|
||||
// (e.g. two `orca node list` invocations racing on the same shell)
|
||||
// wait up to 5s for the writer instead of failing immediately with
|
||||
// SQLITE_BUSY. SetMaxOpenConns(1) serializes the connections so the
|
||||
// busy_timeout is rarely needed but keeps the cache durable under
|
||||
// contention.
|
||||
db, err := sql.Open("sqlite", path+"?_pragma=journal_mode(WAL)&_pragma=busy_timeout(5000)")
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("open cache sqlite: %w", err)
|
||||
}
|
||||
db.SetMaxOpenConns(1)
|
||||
if err := db.Ping(); err != nil {
|
||||
_ = db.Close()
|
||||
return nil, fmt.Errorf("ping cache sqlite: %w", err)
|
||||
}
|
||||
// REQ-158 / P09 T4: enforce 0600 on the cache DB file (SQLite
|
||||
// creates it at umask, typically 0644). Match store.Open which
|
||||
// chmods after open+ping (the file exists at this point). Non-fatal
|
||||
// if chmod fails (e.g. the DB is at a path we don't own).
|
||||
_ = os.Chmod(path, 0o600)
|
||||
const schema = `CREATE TABLE IF NOT EXISTS cache_entries (
|
||||
class TEXT NOT NULL,
|
||||
key TEXT NOT NULL,
|
||||
|
||||
Vendored
+47
@@ -2,6 +2,7 @@ package cache
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
"time"
|
||||
@@ -225,3 +226,49 @@ func BenchmarkCacheHit(b *testing.B) {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
// TestCache_FileMode0600 verifies that the cache DB file is created
|
||||
// with mode 0600 (not the default umask 0644) (REQ-158, P09 T4).
|
||||
func TestCache_FileMode0600(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
path := filepath.Join(dir, "orca_cache.db")
|
||||
c, err := Open(path)
|
||||
if err != nil {
|
||||
t.Fatalf("open: %v", err)
|
||||
}
|
||||
defer c.Close()
|
||||
|
||||
info, err := os.Stat(path)
|
||||
if err != nil {
|
||||
t.Fatalf("stat cache db: %v", err)
|
||||
}
|
||||
got := info.Mode().Perm()
|
||||
if got != 0o600 {
|
||||
t.Errorf("cache db mode = %04o, want 0600", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestCache_FileMode0600DefaultPath verifies that the cache DB at the
|
||||
// default path (ORCA_HOME) also gets 0600 (REQ-158, P09 T4).
|
||||
func TestCache_FileMode0600DefaultPath(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
t.Setenv("ORCA_HOME", dir)
|
||||
c, err := Open("")
|
||||
if err != nil {
|
||||
t.Fatalf("open default path: %v", err)
|
||||
}
|
||||
defer c.Close()
|
||||
|
||||
// The default path is paths.CacheDB() which is under ORCA_HOME.
|
||||
// Find the db file.
|
||||
dbPath := filepath.Join(dir, "orca_cache.db")
|
||||
info, err := os.Stat(dbPath)
|
||||
if err != nil {
|
||||
t.Fatalf("stat cache db at %s: %v", dbPath, err)
|
||||
}
|
||||
got := info.Mode().Perm()
|
||||
if got != 0o600 {
|
||||
t.Errorf("cache db mode = %04o, want 0600", got)
|
||||
}
|
||||
}
|
||||
|
||||
+8
-25
@@ -175,32 +175,15 @@ func lockACL() (func(), error) {
|
||||
return security.Flock(paths.ACLPath() + ".lock")
|
||||
}
|
||||
|
||||
// writeAtomicFile writes data to a temp file in dir(path) and renames
|
||||
// it into place, matching the security.WriteAtomic pattern (P02 keeps
|
||||
// a local copy to avoid importing internal/security into the CLI).
|
||||
// writeAtomicFile writes data atomically (REQ-156, P07 T9).
|
||||
// Previously a local copy of the temp+chmod+rename pattern (P02 kept a
|
||||
// local copy to avoid importing internal/security); it lacked fsync,
|
||||
// so a crash between write and rename could promote a partially-durable
|
||||
// file. Now a thin wrapper around the canonical security.WriteAtomic
|
||||
// (temp + chmod + fsync + rename) so all CLI atomic writes share one
|
||||
// fsync-correct implementation.
|
||||
func writeAtomicFile(path string, data []byte, mode os.FileMode) error {
|
||||
dir := filepath.Dir(path)
|
||||
tmp, err := os.CreateTemp(dir, ".acl-tmp-*")
|
||||
if err != nil {
|
||||
return fmt.Errorf("create temp: %w", err)
|
||||
}
|
||||
tmpName := tmp.Name()
|
||||
defer func() { _ = os.Remove(tmpName) }()
|
||||
if _, err := tmp.Write(data); err != nil {
|
||||
_ = tmp.Close()
|
||||
return fmt.Errorf("write temp: %w", err)
|
||||
}
|
||||
if err := tmp.Chmod(mode); err != nil {
|
||||
_ = tmp.Close()
|
||||
return fmt.Errorf("chmod temp: %w", err)
|
||||
}
|
||||
if err := tmp.Close(); err != nil {
|
||||
return fmt.Errorf("close temp: %w", err)
|
||||
}
|
||||
if err := os.Rename(tmpName, path); err != nil {
|
||||
return fmt.Errorf("rename temp: %w", err)
|
||||
}
|
||||
return nil
|
||||
return security.WriteAtomic(path, mode, data)
|
||||
}
|
||||
|
||||
var aclGrantCmd = &cobra.Command{
|
||||
|
||||
@@ -12,6 +12,8 @@ package cli
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"time"
|
||||
|
||||
"github.com/spf13/cobra"
|
||||
@@ -29,6 +31,31 @@ var (
|
||||
restoreDryRun bool
|
||||
)
|
||||
|
||||
// acquireBackupLock atomically creates an exclusive lock file at
|
||||
// paths.ClusterDir()/backup.lock (REQ-156, P07 T4). Returns a release
|
||||
// function that MUST be deferred (it removes the lock file). If the
|
||||
// lock file already exists, returns an error "backup already in
|
||||
// progress" — preventing two concurrent `orca backup` invocations
|
||||
// from racing on the same ORCA_HOME (two tarballs being written from
|
||||
// the same source tree could produce inconsistent archives). O_CREATE
|
||||
// |O_EXCL is atomic under POSIX.
|
||||
func acquireBackupLock() (func(), error) {
|
||||
lockPath := filepath.Join(paths.ClusterDir(), "backup.lock")
|
||||
if err := os.MkdirAll(filepath.Dir(lockPath), 0o755); err != nil {
|
||||
return nil, fmt.Errorf("create cluster dir for backup lock: %w", err)
|
||||
}
|
||||
f, err := os.OpenFile(lockPath, os.O_CREATE|os.O_EXCL|os.O_WRONLY, 0o600)
|
||||
if err != nil {
|
||||
if os.IsExist(err) {
|
||||
return nil, fmt.Errorf("backup already in progress (lock file %s exists; remove it if stale)", lockPath)
|
||||
}
|
||||
return nil, fmt.Errorf("acquire backup lock: %w", err)
|
||||
}
|
||||
_, _ = f.WriteString(fmt.Sprintf("pid=%d started=%s\n", os.Getpid(), time.Now().UTC().Format(time.RFC3339)))
|
||||
_ = f.Close()
|
||||
return func() { _ = os.Remove(lockPath) }, nil
|
||||
}
|
||||
|
||||
var backupCmd = &cobra.Command{
|
||||
Use: "backup",
|
||||
Short: "Create a signed tar.gz backup of ORCA_HOME",
|
||||
@@ -44,6 +71,14 @@ written to --out; the hex-encoded signature to --out + ".sig".`,
|
||||
if err != nil {
|
||||
return fmt.Errorf("load master key: %w", err)
|
||||
}
|
||||
// REQ-156 / P07 T4: acquire an exclusive backup lock so two
|
||||
// concurrent `orca backup` invocations don't race on the same
|
||||
// ORCA_HOME (producing interleaved / inconsistent archives).
|
||||
backupRelease, err := acquireBackupLock()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer backupRelease()
|
||||
out := backupOutPath
|
||||
if out == "" {
|
||||
ts := time.Now().UTC().Format("20060102-150405")
|
||||
|
||||
@@ -101,6 +101,27 @@ func cachePutList(class, key string, list any, ttl time.Duration) {
|
||||
cachePopulate(class, key, val, ttl)
|
||||
}
|
||||
|
||||
// cacheInvalidate drops all entries for the given cache class
|
||||
// (REQ-156, P07 T5). It is called after write operations (node
|
||||
// join/leave, ns create/delete, job run/stop) so the very next read
|
||||
// does not surface a stale cached list. Errors are logged but never
|
||||
// returned — a failed invalidation must not break the write command
|
||||
// (the cache entry will simply expire at its TTL).
|
||||
func cacheInvalidate(class string) {
|
||||
if !cacheAvailable() {
|
||||
return
|
||||
}
|
||||
c, err := cache.Open(paths.CacheDB())
|
||||
if err != nil {
|
||||
slog.Warn("cache: open failed during invalidate", "class", class, "err", err)
|
||||
return
|
||||
}
|
||||
defer c.Close()
|
||||
if err := c.Invalidate(class); err != nil {
|
||||
slog.Warn("cache: invalidate failed", "class", class, "err", err)
|
||||
}
|
||||
}
|
||||
|
||||
// Per-class TTLs (P00-T2).
|
||||
const (
|
||||
cacheNodeTTL = 30 * time.Second
|
||||
|
||||
@@ -0,0 +1,422 @@
|
||||
package cli
|
||||
|
||||
// concurrency_test.go covers the REQ-156 / P07 concurrency-safety
|
||||
// fixes:
|
||||
//
|
||||
// - T11: concurrent `secrets set` on the same namespace preserves all
|
||||
// keys (the flock serializes the read-modify-write so no key is
|
||||
// lost to a clobbering second writer).
|
||||
// - T12: a second `orca upgrade` invoked while the first is running
|
||||
// is rejected with "upgrade already in progress".
|
||||
// - T13: cache invalidation read-after-write - `node join` followed
|
||||
// by an immediate `node list` (with a populated stale cache) shows
|
||||
// the new node, not the stale cached list.
|
||||
// - T14: (in internal/webauthn) concurrent BeginRegistration does
|
||||
// not panic / race on the session map.
|
||||
//
|
||||
// These tests complement the per-fix unit tests in the relevant
|
||||
// _test.go files; they specifically exercise the cross-cutting
|
||||
// concurrency invariants the milestone hardens.
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"git.cloudinit.dev/coreci/orca/internal/cache"
|
||||
"git.cloudinit.dev/coreci/orca/internal/paths"
|
||||
"git.cloudinit.dev/coreci/orca/internal/secrets"
|
||||
)
|
||||
|
||||
// runCLI is a helper that resets root flags, wires a fresh output
|
||||
// buffer, sets the given args, and runs rootCmd. Returns the captured
|
||||
// output. The buffer must be wired AFTER resetRootFlags (which sets
|
||||
// its own buffer).
|
||||
func runCLI(t *testing.T, args ...string) (string, error) {
|
||||
t.Helper()
|
||||
resetRootFlags(t)
|
||||
var buf bytes.Buffer
|
||||
rootCmd.SetOut(&buf)
|
||||
rootCmd.SetErr(&buf)
|
||||
rootCmd.SetArgs(args)
|
||||
err := rootCmd.Execute()
|
||||
return buf.String(), err
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// T11: concurrent secrets set preserves all keys
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
// TestSecretsConcurrentSetPreservesAllKeys runs 5 concurrent
|
||||
// `orca secrets set` invocations against the SAME namespace, each
|
||||
// setting a distinct key. Without the flock (P07 T2) the second writer
|
||||
// would load-then-save and clobber the first, losing a key. With the
|
||||
// flock all 5 keys must be present afterward.
|
||||
//
|
||||
// The cobra rootCmd is a package global and is NOT goroutine-safe
|
||||
// (shared flag state), so we drive the secrets-set RunE body directly
|
||||
// under real concurrency. This exercises the lockNSSecrets flock +
|
||||
// loadMasterAndNSSecrets + saveNSSecrets path that the RunE uses.
|
||||
func TestSecretsConcurrentSetPreservesAllKeys(t *testing.T) {
|
||||
ns := "concsetns"
|
||||
setupSecretsTestEnv(t, ns)
|
||||
|
||||
const n = 5
|
||||
keys := make([]string, n)
|
||||
for i := 0; i < n; i++ {
|
||||
keys[i] = fmt.Sprintf("KEY_%d", i)
|
||||
}
|
||||
|
||||
var wg sync.WaitGroup
|
||||
errs := make([]error, n)
|
||||
for i := 0; i < n; i++ {
|
||||
wg.Add(1)
|
||||
go func(idx int) {
|
||||
defer wg.Done()
|
||||
// Replicate the secretsSetCmd RunE body under real
|
||||
// concurrency: lock -> load -> mutate -> save. The lock
|
||||
// serializes the read-modify-write so concurrent sets do
|
||||
// not clobber each other.
|
||||
release, err := lockNSSecrets(ns)
|
||||
if err != nil {
|
||||
errs[idx] = fmt.Errorf("lock: %w", err)
|
||||
return
|
||||
}
|
||||
defer release()
|
||||
nsKey, lines, err := loadMasterAndNSSecrets(ns)
|
||||
if err != nil {
|
||||
errs[idx] = err
|
||||
return
|
||||
}
|
||||
defer secrets.ZeroKey(nsKey)
|
||||
key := keys[idx]
|
||||
value := fmt.Sprintf("value_%d", idx)
|
||||
newLine := key + "=" + value
|
||||
j := findKeyIndex(lines, key)
|
||||
if j >= 0 {
|
||||
lines[j] = newLine
|
||||
} else {
|
||||
lines = append(lines, newLine)
|
||||
}
|
||||
errs[idx] = saveNSSecrets(ns, nsKey, lines)
|
||||
}(i)
|
||||
}
|
||||
wg.Wait()
|
||||
|
||||
for i, err := range errs {
|
||||
if err != nil {
|
||||
t.Fatalf("goroutine %d: %v", i, err)
|
||||
}
|
||||
}
|
||||
|
||||
// All 5 keys must be present.
|
||||
out, err := runCLI(t, "secrets", "list", ns)
|
||||
if err != nil {
|
||||
t.Fatalf("secrets list: %v", err)
|
||||
}
|
||||
for _, k := range keys {
|
||||
if !strings.Contains(out, k) {
|
||||
t.Errorf("key %q missing after concurrent set (flock did not serialize): %s", k, out)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestSecretsConcurrentSetViaCLI is the cobra-driven variant. cobra's
|
||||
// rootCmd is not goroutine-safe (shared flag globals), so we serialize
|
||||
// the Execute() calls. This still exercises the flock because the
|
||||
// load+save happens inside RunE. Confirms the CLI path itself (with
|
||||
// flock) does not lose keys under repeated serial sets.
|
||||
func TestSecretsConcurrentSetViaCLI(t *testing.T) {
|
||||
ns := "conccli"
|
||||
setupSecretsTestEnv(t, ns)
|
||||
|
||||
const n = 5
|
||||
for i := 0; i < n; i++ {
|
||||
if _, err := runCLI(t, "secrets", "set", ns, fmt.Sprintf("K_%d=v_%d", i, i)); err != nil {
|
||||
t.Fatalf("secrets set %d: %v", i, err)
|
||||
}
|
||||
}
|
||||
out, err := runCLI(t, "secrets", "list", ns)
|
||||
if err != nil {
|
||||
t.Fatalf("secrets list: %v", err)
|
||||
}
|
||||
for i := 0; i < n; i++ {
|
||||
k := fmt.Sprintf("K_%d", i)
|
||||
if !strings.Contains(out, k) {
|
||||
t.Errorf("key %q missing after serial CLI sets: %s", k, out)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// T12: concurrent upgrade rejection
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
// TestUpgradeConcurrentLockRejected verifies that a second upgrade
|
||||
// invocation while the first holds the upgrade.lock is rejected with
|
||||
// "upgrade already in progress".
|
||||
func TestUpgradeConcurrentLockRejected(t *testing.T) {
|
||||
setupUpgradeTest(t)
|
||||
resetUpgradeFlags()
|
||||
|
||||
// Manually create the upgrade.lock as if a first upgrade is in
|
||||
// progress (the lock file content is just diagnostic; its
|
||||
// EXISTENCE is what blocks the second caller via O_CREATE|O_EXCL).
|
||||
lockPath := filepath.Join(paths.ClusterDir(), "upgrade.lock")
|
||||
if err := os.MkdirAll(filepath.Dir(lockPath), 0o755); err != nil {
|
||||
t.Fatalf("mkdir cluster: %v", err)
|
||||
}
|
||||
if err := os.WriteFile(lockPath, []byte("pid=999 started=2026-01-01T00:00:00Z\n"), 0o600); err != nil {
|
||||
t.Fatalf("write lock: %v", err)
|
||||
}
|
||||
defer os.Remove(lockPath)
|
||||
|
||||
// A dry-run upgrade must now be rejected because the lock exists.
|
||||
_, err := runCLI(t, "upgrade", "--to", "v0.11.0", "--dry-run")
|
||||
if err == nil {
|
||||
t.Fatal("upgrade with stale lock should fail, got nil")
|
||||
}
|
||||
if !strings.Contains(err.Error(), "upgrade already in progress") {
|
||||
t.Errorf("unexpected error: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestUpgradeLockReleasedOnSuccess verifies the upgrade.lock is
|
||||
// removed after a successful (dry-run) upgrade so a subsequent upgrade
|
||||
// is not blocked by a stale lock.
|
||||
func TestUpgradeLockReleasedOnSuccess(t *testing.T) {
|
||||
setupUpgradeTest(t)
|
||||
setupUpgradeTestWithMocks(t)
|
||||
rootCmd.SetArgs([]string{"upgrade", "--to", "v0.11.0", "--dry-run"})
|
||||
if err := rootCmd.Execute(); err != nil {
|
||||
t.Fatalf("upgrade dry-run: %v", err)
|
||||
}
|
||||
lockPath := filepath.Join(paths.ClusterDir(), "upgrade.lock")
|
||||
if _, err := os.Stat(lockPath); err == nil {
|
||||
t.Errorf("upgrade.lock still exists after successful dry-run (not released): %s", lockPath)
|
||||
}
|
||||
}
|
||||
|
||||
// TestUpgradeLockReleasedOnError verifies the lock is released even
|
||||
// when the upgrade fails mid-run (the defer in runUpgrade covers the
|
||||
// error path).
|
||||
func TestUpgradeLockReleasedOnError(t *testing.T) {
|
||||
setupUpgradeTest(t)
|
||||
setupUpgradeTestWithMocks(t)
|
||||
// Force a failure: --to with a version that triggers a cutover
|
||||
// whose verification fails. The runner reports :443 (cutover
|
||||
// needed) and the http check returns 502 (verification fail).
|
||||
runner := &mockUpgradeRunner{
|
||||
outputs: map[string][]byte{
|
||||
"ss -tlnp": []byte(":443"),
|
||||
},
|
||||
}
|
||||
upgradeRunnerOverride = runner
|
||||
httpClientOverride = func(url string) (int, error) { return 502, nil }
|
||||
rootCmd.SetArgs([]string{"upgrade", "--to", "v0.11.0"})
|
||||
_ = rootCmd.Execute() // expected to fail
|
||||
lockPath := filepath.Join(paths.ClusterDir(), "upgrade.lock")
|
||||
if _, err := os.Stat(lockPath); err == nil {
|
||||
t.Errorf("upgrade.lock still exists after failed upgrade (not released on error): %s", lockPath)
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// T13: cache invalidation read-after-write
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
// TestCacheInvalidationNodeJoinReadAfterWrite verifies that after
|
||||
// `node join` invalidates the `nodes` cache class, an immediate
|
||||
// `node list` (which would otherwise serve a STALE cached list) shows
|
||||
// the just-joined node.
|
||||
//
|
||||
// Setup: populate the cache with a stale nodes list (missing the new
|
||||
// node). Without T5's invalidation, the second `node list` would serve
|
||||
// the stale list and the new node would be invisible until the TTL
|
||||
// expired. With T5, the join invalidates the class and the list
|
||||
// re-reads from the DB.
|
||||
func TestCacheInvalidationNodeJoinReadAfterWrite(t *testing.T) {
|
||||
_, cleanup := initTestEnv(t)
|
||||
defer cleanup()
|
||||
|
||||
// Seed the cache with a stale nodes list (a sentinel node that
|
||||
// does NOT exist in the DB). The TTL is long so it would be
|
||||
// served on a subsequent list without invalidation.
|
||||
c, err := cache.Open(paths.CacheDB())
|
||||
if err != nil {
|
||||
t.Fatalf("open cache: %v", err)
|
||||
}
|
||||
stale := `[{"id":"stale-id","name":"stale-node","address":"10.0.0.99:8443","state":"ready"}]`
|
||||
if err := c.Set(cacheNodeClass, cacheListKey, []byte(stale), 10*time.Minute); err != nil {
|
||||
t.Fatalf("set stale cache: %v", err)
|
||||
}
|
||||
c.Close()
|
||||
|
||||
// Confirm the stale entry is served by a fresh list (proving the
|
||||
// cache is populated and would be hit).
|
||||
staleOut, err := runCLI(t, "node", "list")
|
||||
if err != nil {
|
||||
t.Fatalf("stale node list: %v", err)
|
||||
}
|
||||
if !strings.Contains(staleOut, "stale-node") {
|
||||
t.Fatalf("precondition: stale cache not served: %s", staleOut)
|
||||
}
|
||||
|
||||
// Join a real node. T5 invalidates the `nodes` cache class.
|
||||
if _, err := runCLI(t, "node", "join", "--name", "freshnode", "--addr", "10.0.0.42:8443"); err != nil {
|
||||
t.Fatalf("node join: %v", err)
|
||||
}
|
||||
|
||||
// Immediate list: the stale sentinel must be GONE (invalidated)
|
||||
// and the real fresh node must be present (read from the DB).
|
||||
out, err := runCLI(t, "node", "list")
|
||||
if err != nil {
|
||||
t.Fatalf("node list after join: %v", err)
|
||||
}
|
||||
if strings.Contains(out, "stale-node") {
|
||||
t.Errorf("stale cache still served after join (invalidation missing): %s", out)
|
||||
}
|
||||
if !strings.Contains(out, "freshnode") {
|
||||
t.Errorf("fresh node missing from list after join (cache not re-read): %s", out)
|
||||
}
|
||||
}
|
||||
|
||||
// TestCacheInvalidationNSCreateReadAfterWrite is the ns variant: a
|
||||
// stale `namespaces` cache is invalidated by `ns create` so the next
|
||||
// `ns list` shows the new namespace.
|
||||
func TestCacheInvalidationNSCreateReadAfterWrite(t *testing.T) {
|
||||
root := t.TempDir()
|
||||
t.Setenv("ORCA_HOME", root)
|
||||
writeDefaultsNS(t, root)
|
||||
|
||||
// Seed a stale namespaces cache containing only _defaults.
|
||||
c, err := cache.Open(paths.CacheDB())
|
||||
if err != nil {
|
||||
t.Fatalf("open cache: %v", err)
|
||||
}
|
||||
stale := `[{"name":"_defaults","path":"` + filepath.Join(root, "_defaults") + `","default":true}]`
|
||||
if err := c.Set(cacheNamespaceClass, cacheListKey, []byte(stale), 10*time.Minute); err != nil {
|
||||
t.Fatalf("set stale: %v", err)
|
||||
}
|
||||
c.Close()
|
||||
|
||||
// Confirm stale served.
|
||||
resetRootFlags(t)
|
||||
resetNSFlags()
|
||||
staleOut, err := runCLI(t, "ns", "list")
|
||||
if err != nil {
|
||||
t.Fatalf("stale ns list: %v", err)
|
||||
}
|
||||
if !strings.Contains(staleOut, "_defaults") {
|
||||
t.Fatalf("precondition: stale ns cache not served: %s", staleOut)
|
||||
}
|
||||
|
||||
// Create a new namespace. T5 invalidates the `namespaces` cache.
|
||||
resetRootFlags(t)
|
||||
resetNSFlags()
|
||||
if _, err := runCLI(t, "ns", "create", "newns"); err != nil {
|
||||
t.Fatalf("ns create: %v", err)
|
||||
}
|
||||
|
||||
// Immediate list: must show the new namespace (read from disk,
|
||||
// not the stale cache).
|
||||
resetRootFlags(t)
|
||||
resetNSFlags()
|
||||
out, err := runCLI(t, "ns", "list")
|
||||
if err != nil {
|
||||
t.Fatalf("ns list after create: %v", err)
|
||||
}
|
||||
if !strings.Contains(out, "newns") {
|
||||
t.Errorf("new namespace missing from list after create (cache not invalidated/re-read): %s", out)
|
||||
}
|
||||
}
|
||||
|
||||
// TestCacheInvalidationJobRunReadAfterWrite verifies `job run`
|
||||
// invalidates the `jobs` cache so a stale cached job list is not
|
||||
// served after a new job runs.
|
||||
func TestCacheInvalidationJobRunReadAfterWrite(t *testing.T) {
|
||||
_, cleanup := initTestEnv(t)
|
||||
defer cleanup()
|
||||
|
||||
// Seed a stale jobs cache (a sentinel job that does not exist).
|
||||
c, err := cache.Open(paths.CacheDB())
|
||||
if err != nil {
|
||||
t.Fatalf("open cache: %v", err)
|
||||
}
|
||||
stale := `[{"id":"stale-job","name":"stale","status":"complete","exit_code":0}]`
|
||||
if err := c.Set(cacheJobClass, cacheListKey, []byte(stale), 10*time.Minute); err != nil {
|
||||
t.Fatalf("set stale: %v", err)
|
||||
}
|
||||
c.Close()
|
||||
|
||||
// Confirm stale served.
|
||||
staleOut, err := runCLI(t, "job", "list")
|
||||
if err != nil {
|
||||
t.Fatalf("stale job list: %v", err)
|
||||
}
|
||||
if !strings.Contains(staleOut, "stale") {
|
||||
t.Fatalf("precondition: stale job cache not served: %s", staleOut)
|
||||
}
|
||||
|
||||
// Write a job spec and run it. T5 invalidates the `jobs` cache.
|
||||
specDir := t.TempDir()
|
||||
specPath := filepath.Join(specDir, "job.md")
|
||||
specBody := "---\n" +
|
||||
"kind: Job\n" +
|
||||
"name: cacheinv-job\n" +
|
||||
"runtime:\n" +
|
||||
" one_of: process\n" +
|
||||
" command: /bin/true\n" +
|
||||
"---\n# cacheinv\n\nRuns /bin/true.\n"
|
||||
if err := os.WriteFile(specPath, []byte(specBody), 0o644); err != nil {
|
||||
t.Fatalf("write spec: %v", err)
|
||||
}
|
||||
if _, err := runCLI(t, "job", "run", specPath); err != nil {
|
||||
t.Fatalf("job run: %v", err)
|
||||
}
|
||||
|
||||
// Immediate list: the stale sentinel must be gone; the real job
|
||||
// must be present (read from the DB).
|
||||
out, err := runCLI(t, "job", "list")
|
||||
if err != nil {
|
||||
t.Fatalf("job list after run: %v", err)
|
||||
}
|
||||
if strings.Contains(out, "stale-job") {
|
||||
t.Errorf("stale job cache still served after run (invalidation missing): %s", out)
|
||||
}
|
||||
if !strings.Contains(out, "cacheinv-job") {
|
||||
t.Errorf("new job missing from list after run (cache not re-read): %s", out)
|
||||
}
|
||||
}
|
||||
|
||||
// TestCacheInvalidateHelperDirectly is a small unit test for the
|
||||
// cacheInvalidate helper itself: it confirms a populated class is
|
||||
// empty after the helper runs.
|
||||
func TestCacheInvalidateHelperDirectly(t *testing.T) {
|
||||
_, cleanup := initTestEnv(t)
|
||||
defer cleanup()
|
||||
c, err := cache.Open(paths.CacheDB())
|
||||
if err != nil {
|
||||
t.Fatalf("open: %v", err)
|
||||
}
|
||||
if err := c.Set(cacheNodeClass, cacheListKey, []byte("x"), 0); err != nil {
|
||||
t.Fatalf("set: %v", err)
|
||||
}
|
||||
c.Close()
|
||||
|
||||
cacheInvalidate(cacheNodeClass)
|
||||
|
||||
c2, err := cache.Open(paths.CacheDB())
|
||||
if err != nil {
|
||||
t.Fatalf("reopen: %v", err)
|
||||
}
|
||||
defer c2.Close()
|
||||
if _, _, err := c2.Get(cacheNodeClass, cacheListKey); err == nil {
|
||||
t.Errorf("nodes/list still present after cacheInvalidate")
|
||||
}
|
||||
}
|
||||
+75
-1
@@ -384,7 +384,81 @@ func checkOIDCHealth(ctx context.Context) []oidcCheckResult {
|
||||
return results
|
||||
}
|
||||
|
||||
// doctorDBRetentionCmd implements `orca doctor db-retention` (REQ-158,
|
||||
// P09 T2). Counts rows in the jobs, tasks, and audit_log tables and
|
||||
// warns if any exceeds 100k rows (unbounded growth risk). Suggests
|
||||
// `orca backup` + manual cleanup.
|
||||
var doctorDBRetentionCmd = &cobra.Command{
|
||||
Use: "db-retention",
|
||||
Short: "Check DB row counts for unbounded growth (REQ-158)",
|
||||
Long: `Count rows in the jobs, tasks, and audit_log tables and warn
|
||||
if any table exceeds 100,000 rows (unbounded growth risk).
|
||||
|
||||
Large tables degrade query performance and inflate backup size. Run
|
||||
'orca backup' to capture a snapshot, then prune old rows manually
|
||||
(e.g. DELETE FROM tasks WHERE created_at < <cutoff>).
|
||||
|
||||
Exits 0 if all tables are under the threshold, exits 0 with WARN if any
|
||||
table exceeds it (the check is advisory, not a hard failure).`,
|
||||
Args: cobra.NoArgs,
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
ctx, cancel := context.WithTimeout(cmd.Context(), 10*time.Second)
|
||||
defer cancel()
|
||||
|
||||
db, closer, err := openDB()
|
||||
if err != nil {
|
||||
return fmt.Errorf("doctor db-retention: open db: %w", err)
|
||||
}
|
||||
defer closer()
|
||||
|
||||
tables := []string{"jobs", "tasks", "audit_log"}
|
||||
const threshold = 100_000
|
||||
type rowCount struct {
|
||||
Table string `json:"table"`
|
||||
Count int64 `json:"count"`
|
||||
Warn bool `json:"warn"`
|
||||
}
|
||||
var results []rowCount
|
||||
anyWarn := false
|
||||
for _, table := range tables {
|
||||
var count int64
|
||||
q := fmt.Sprintf("SELECT COUNT(*) FROM %s", table)
|
||||
if err := db.QueryRowContext(ctx, q).Scan(&count); err != nil {
|
||||
return fmt.Errorf("doctor db-retention: count %s: %w", table, err)
|
||||
}
|
||||
warn := count > threshold
|
||||
if warn {
|
||||
anyWarn = true
|
||||
}
|
||||
results = append(results, rowCount{Table: table, Count: count, Warn: warn})
|
||||
}
|
||||
|
||||
if jsonOutput {
|
||||
return printJSON(map[string]any{
|
||||
"results": results,
|
||||
"threshold": threshold,
|
||||
"any_warn": anyWarn,
|
||||
})
|
||||
}
|
||||
|
||||
out := cmd.OutOrStdout()
|
||||
for _, r := range results {
|
||||
status := "ok"
|
||||
if r.Warn {
|
||||
status = "WARN"
|
||||
}
|
||||
fmt.Fprintf(out, "%-12s %-5s %d rows (threshold: %d)\n", r.Table, status, r.Count, threshold)
|
||||
}
|
||||
if anyWarn {
|
||||
fmt.Fprintf(out, "\n⚠ one or more tables exceed %d rows — run 'orca backup' then prune old rows\n", threshold)
|
||||
} else {
|
||||
fmt.Fprintln(out, "\n✓ all tables under retention threshold")
|
||||
}
|
||||
return nil
|
||||
},
|
||||
}
|
||||
|
||||
func init() {
|
||||
doctorCmd.AddCommand(doctorCertCmd, doctorNetworkCmd, doctorDBCmd, doctorOSCmd, doctorProxmoxCmd, noOrcaOnServerCmd, doctorNftCmd, doctorAuditCmd, doctorModesCmd, doctorOIDCCmd)
|
||||
doctorCmd.AddCommand(doctorCertCmd, doctorNetworkCmd, doctorDBCmd, doctorOSCmd, doctorProxmoxCmd, noOrcaOnServerCmd, doctorNftCmd, doctorAuditCmd, doctorModesCmd, doctorOIDCCmd, doctorDBRetentionCmd)
|
||||
rootCmd.AddCommand(doctorCmd)
|
||||
}
|
||||
|
||||
@@ -2,9 +2,14 @@ package cli
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"encoding/json"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"git.cloudinit.dev/coreci/orca/internal/certpaths"
|
||||
"git.cloudinit.dev/coreci/orca/internal/store"
|
||||
)
|
||||
|
||||
func TestDoctorText(t *testing.T) {
|
||||
@@ -194,3 +199,92 @@ func TestDoctorProxmoxJSON(t *testing.T) {
|
||||
t.Errorf("doctor proxmox --json missing Name: %v", result)
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
// TestDoctorDBRetention verifies that `orca doctor db-retention` counts
|
||||
// rows in jobs, tasks, and audit_log and warns when a table exceeds
|
||||
// 100k rows (REQ-158, P09 T7).
|
||||
func TestDoctorDBRetention(t *testing.T) {
|
||||
_, cleanup := initTestEnv(t)
|
||||
defer cleanup()
|
||||
if err := runInit(discardWriter{}); err != nil {
|
||||
t.Fatalf("init: %v", err)
|
||||
}
|
||||
|
||||
// Insert 100001 rows into the audit_log table to trigger the warning.
|
||||
// Use a multi-row VALUES insert in batches for speed.
|
||||
db, err := store.Open(certpaths.DBPath())
|
||||
if err != nil {
|
||||
t.Fatalf("open db: %v", err)
|
||||
}
|
||||
defer db.Close()
|
||||
ctx := context.Background()
|
||||
// Build a batch insert: 500 rows per INSERT in a transaction.
|
||||
// SQLite handles this much faster than 100k individual inserts.
|
||||
const totalRows = 100001
|
||||
const batchSize = 500
|
||||
inserted := 0
|
||||
for inserted < totalRows {
|
||||
remaining := totalRows - inserted
|
||||
batch := batchSize
|
||||
if remaining < batch {
|
||||
batch = remaining
|
||||
}
|
||||
var placeholders strings.Builder
|
||||
var args []any
|
||||
for j := 0; j < batch; j++ {
|
||||
if j > 0 {
|
||||
placeholders.WriteString(",")
|
||||
}
|
||||
placeholders.WriteString("(?, 'test', 'test.action', 'test-resource', 'success')")
|
||||
args = append(args, time.Now().UTC())
|
||||
}
|
||||
q := "INSERT INTO audit_log (timestamp, actor, action, resource, result) VALUES " + placeholders.String()
|
||||
if _, err := db.ExecContext(ctx, q, args...); err != nil {
|
||||
t.Fatalf("batch insert at offset %d: %v", inserted, err)
|
||||
}
|
||||
inserted += batch
|
||||
}
|
||||
|
||||
resetRootFlags(t)
|
||||
var buf bytes.Buffer
|
||||
rootCmd.SetOut(&buf)
|
||||
rootCmd.SetErr(&buf)
|
||||
rootCmd.SetArgs([]string{"doctor", "db-retention"})
|
||||
if err := rootCmd.Execute(); err != nil {
|
||||
t.Fatalf("doctor db-retention: %v", err)
|
||||
}
|
||||
out := buf.String()
|
||||
if !strings.Contains(out, "audit_log") {
|
||||
t.Errorf("output missing audit_log table: %s", out)
|
||||
}
|
||||
if !strings.Contains(out, "WARN") {
|
||||
t.Errorf("output should contain WARN for audit_log exceeding threshold: %s", out)
|
||||
}
|
||||
if !strings.Contains(out, "backup") {
|
||||
t.Errorf("output should suggest 'orca backup': %s", out)
|
||||
}
|
||||
}
|
||||
|
||||
// TestDoctorDBRetentionNoWarn verifies that with a small DB no warning
|
||||
// is emitted (REQ-158, P09 T7).
|
||||
func TestDoctorDBRetentionNoWarn(t *testing.T) {
|
||||
_, cleanup := initTestEnv(t)
|
||||
defer cleanup()
|
||||
if err := runInit(discardWriter{}); err != nil {
|
||||
t.Fatalf("init: %v", err)
|
||||
}
|
||||
|
||||
resetRootFlags(t)
|
||||
var buf bytes.Buffer
|
||||
rootCmd.SetOut(&buf)
|
||||
rootCmd.SetErr(&buf)
|
||||
rootCmd.SetArgs([]string{"doctor", "db-retention"})
|
||||
if err := rootCmd.Execute(); err != nil {
|
||||
t.Fatalf("doctor db-retention: %v", err)
|
||||
}
|
||||
out := buf.String()
|
||||
if strings.Contains(out, "WARN") {
|
||||
t.Errorf("output should NOT contain WARN for small DB: %s", out)
|
||||
}
|
||||
}
|
||||
|
||||
+28
-6
@@ -5,6 +5,7 @@ import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"log/slog"
|
||||
"net"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
@@ -48,12 +49,17 @@ func drainExecFromCtx(_ context.Context) (drainExecer, error) {
|
||||
// Address carries host:8443. We always target SSH port 22 unless the
|
||||
// node's Address already encodes a non-daemon port. The local node
|
||||
// (Name=="localhost") is contacted at "localhost:22".
|
||||
//
|
||||
// REQ-157 / P08 T5: uses net.JoinHostPort for proper IPv6 bracketing
|
||||
// (e.g. "fd00::1" + "22" -> "[fd00::1]:22"). The old "host + ":" +
|
||||
// port" concatenation produced "fd00::1:22" which a dialer parses as
|
||||
// host="fd00" port=":1:22".
|
||||
func peerAddrForNode(n *model.Node) string {
|
||||
if n == nil {
|
||||
return ""
|
||||
}
|
||||
if h, p, ok := splitHostPort(n.Address); ok && p != "" && p != "8443" {
|
||||
return h + ":" + p
|
||||
return net.JoinHostPort(h, p)
|
||||
}
|
||||
host := n.Name
|
||||
if h, _, ok := splitHostPort(n.Address); ok && h != "" && h != "localhost" {
|
||||
@@ -62,15 +68,31 @@ func peerAddrForNode(n *model.Node) string {
|
||||
if host == "" {
|
||||
host = n.Name
|
||||
}
|
||||
return host + ":22"
|
||||
return net.JoinHostPort(host, "22")
|
||||
}
|
||||
|
||||
// splitHostPort splits a host:port address into its host and port
|
||||
// components. It uses net.SplitHostPort for proper IPv6 bracketing
|
||||
// (e.g. "[fd00::1]:8443" -> "fd00::1", "8443"). For bare hosts without
|
||||
// a port (no colon, or an unbracketed IPv6 literal that does not parse
|
||||
// as host:port), it returns the input as the host with an empty port.
|
||||
func splitHostPort(addr string) (string, string, bool) {
|
||||
idx := strings.LastIndex(addr, ":")
|
||||
if idx < 0 {
|
||||
return addr, "", false
|
||||
host, port, err := net.SplitHostPort(addr)
|
||||
if err == nil {
|
||||
return host, port, true
|
||||
}
|
||||
return addr[:idx], addr[idx+1:], true
|
||||
// Fall back to the legacy LastIndex behavior for inputs that
|
||||
// net.SplitHostPort rejects (e.g. bare "localhost" with no port).
|
||||
if idx := strings.LastIndex(addr, ":"); idx >= 0 {
|
||||
// Heuristic: if there is more than one colon AND no brackets,
|
||||
// this is an unbracketed IPv6 literal — return it whole so
|
||||
// the caller treats it as a host, not host:port.
|
||||
if strings.Count(addr, ":") > 1 && !strings.HasPrefix(addr, "[") {
|
||||
return addr, "", false
|
||||
}
|
||||
return addr[:idx], addr[idx+1:], true
|
||||
}
|
||||
return addr, "", false
|
||||
}
|
||||
|
||||
var (
|
||||
|
||||
+26
-3
@@ -69,6 +69,17 @@ func driftTransportFromCtx() (driftTransport, error) {
|
||||
return sshpush.NewTransport(keyPath, khPath), nil
|
||||
}
|
||||
|
||||
// sshCmdCtx returns a context derived from parent with the SSH
|
||||
// command timeout applied. If d <= 0, the parent is returned unchanged
|
||||
// (no deadline). REQ-157 / P08 T6: gives SSH-driven CLI subcommands a
|
||||
// bounded deadline so a hung peer cannot block forever.
|
||||
func sshCmdCtx(parent context.Context, d time.Duration) (context.Context, context.CancelFunc) {
|
||||
if d <= 0 {
|
||||
return context.WithCancel(parent)
|
||||
}
|
||||
return context.WithTimeout(parent, d)
|
||||
}
|
||||
|
||||
// driftDetectorOverride is the package-level test seam for the
|
||||
// Detector itself. When non-nil it replaces the production detector
|
||||
// (which wraps a driftTransport). Tests set it and restore nil.
|
||||
@@ -202,7 +213,9 @@ blocks txn apply for that namespace (R-020).`,
|
||||
if err != nil {
|
||||
return fmt.Errorf("drift detector: %w", err)
|
||||
}
|
||||
if err := d.Acknowledge(cmd.Context(), peer, path); err != nil {
|
||||
ctx, cancel := sshCmdCtx(cmd.Context(), driftAckTimeout)
|
||||
defer cancel()
|
||||
if err := d.Acknowledge(ctx, peer, path); err != nil {
|
||||
return fmt.Errorf("acknowledge: %w", err)
|
||||
}
|
||||
printResult(fmt.Sprintf("✓ Acknowledged drift on %s for %s", peer, path), map[string]any{
|
||||
@@ -225,7 +238,9 @@ var driftRemediateCmd = &cobra.Command{
|
||||
if err != nil {
|
||||
return fmt.Errorf("drift detector: %w", err)
|
||||
}
|
||||
if err := d.Remediate(cmd.Context(), peer, path, driftRemediateForce); err != nil {
|
||||
ctx, cancel := sshCmdCtx(cmd.Context(), driftRemediateTimeout)
|
||||
defer cancel()
|
||||
if err := d.Remediate(ctx, peer, path, driftRemediateForce); err != nil {
|
||||
if errors.Is(err, drift.ErrCooldown) {
|
||||
printResult(fmt.Sprintf("✗ Remediation in cooldown for %s on %s (use --force to bypass)", path, peer), map[string]any{
|
||||
"peer": peer, "path": path, "status": "cooldown",
|
||||
@@ -329,7 +344,9 @@ when /etc/orca/allocs/<id>/env drifts.`,
|
||||
}
|
||||
unit := fmt.Sprintf("orca-alloc-%s.service", name)
|
||||
restartCmd := fmt.Sprintf("systemctl restart %s", shellQuoteDrift(unit))
|
||||
out, err := transport.Exec(cmd.Context(), peer, restartCmd)
|
||||
ctx, cancel := sshCmdCtx(cmd.Context(), jobRestartTimeout)
|
||||
defer cancel()
|
||||
out, err := transport.Exec(ctx, peer, restartCmd)
|
||||
if err != nil {
|
||||
return fmt.Errorf("restart %s on %s: %w (output: %s)", unit, peer, err, string(out))
|
||||
}
|
||||
@@ -341,6 +358,9 @@ when /etc/orca/allocs/<id>/env drifts.`,
|
||||
}
|
||||
|
||||
var jobRestartPeer string
|
||||
var driftRemediateTimeout time.Duration
|
||||
var driftAckTimeout time.Duration
|
||||
var jobRestartTimeout time.Duration
|
||||
|
||||
func shellQuoteDrift(s string) string {
|
||||
return "'" + strings.ReplaceAll(s, "'", "'\\''") + "'"
|
||||
@@ -351,6 +371,9 @@ func init() {
|
||||
driftWatchCmd.Flags().StringSliceVar(&driftWatchPaths, "paths", nil, "comma-separated glob patterns to watch (default: all)")
|
||||
driftShowCmd.Flags().StringVar(&driftShowPeer, "peer", "", "filter to a single peer host")
|
||||
driftRemediateCmd.Flags().BoolVar(&driftRemediateForce, "force", false, "bypass the cooldown window (C4)")
|
||||
driftRemediateCmd.Flags().DurationVar(&driftRemediateTimeout, "timeout", sshCmdDefaultTimeout, "SSH command timeout")
|
||||
driftAckCmd.Flags().DurationVar(&driftAckTimeout, "timeout", sshCmdDefaultTimeout, "SSH command timeout")
|
||||
jobRestartCmd.Flags().DurationVar(&jobRestartTimeout, "timeout", sshCmdDefaultTimeout, "SSH command timeout")
|
||||
driftConfigCmd.PersistentFlags().StringVar(&driftConfigPath, "config", "", "path to drift config JSON (default: built-in)")
|
||||
jobRestartCmd.Flags().StringVar(&jobRestartPeer, "peer", "", "peer address (host:port) running the allocation")
|
||||
|
||||
|
||||
+150
-9
@@ -2,6 +2,7 @@ package cli
|
||||
|
||||
import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
@@ -14,9 +15,11 @@ import (
|
||||
"github.com/google/uuid"
|
||||
"github.com/spf13/cobra"
|
||||
|
||||
"git.cloudinit.dev/coreci/orca/internal/certpaths"
|
||||
"git.cloudinit.dev/coreci/orca/internal/engine"
|
||||
"git.cloudinit.dev/coreci/orca/internal/jobspec"
|
||||
"git.cloudinit.dev/coreci/orca/internal/model"
|
||||
"git.cloudinit.dev/coreci/orca/internal/sshpush"
|
||||
"git.cloudinit.dev/coreci/orca/internal/store"
|
||||
)
|
||||
|
||||
@@ -44,8 +47,8 @@ var (
|
||||
)
|
||||
|
||||
var jobRunCmd = &cobra.Command{
|
||||
Use: "run <spec.hcl>",
|
||||
Short: "Run a job from an HCL spec file",
|
||||
Use: "run <spec.md>",
|
||||
Short: "Run a job from a markdown spec file",
|
||||
Long: "Submit a job spec, execute its tasks, and persist the result. Use --target to pin to a specific node (overrides bin-packing); --idempotency-key for cross-node dispatch dedupe.",
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
@@ -123,6 +126,9 @@ var jobRunCmd = &cobra.Command{
|
||||
return derr
|
||||
}
|
||||
res.unitPaths = unitPaths
|
||||
// REQ-156 / P07 T5: invalidate the jobs cache (the
|
||||
// dispatch decision records a local job entry).
|
||||
cacheInvalidate(cacheJobClass)
|
||||
if jsonOutput {
|
||||
return printJSON(map[string]any{
|
||||
"status": "deployed",
|
||||
@@ -149,6 +155,10 @@ var jobRunCmd = &cobra.Command{
|
||||
}
|
||||
runErr := exec.Run(ctx, job, workloadToTaskSpecs(spec))
|
||||
logDispatch(res, runErr)
|
||||
// REQ-156 / P07 T5: invalidate the jobs cache so the next
|
||||
// `orca job list` reflects the just-run (or just-failed)
|
||||
// job instead of a stale cached list.
|
||||
cacheInvalidate(cacheJobClass)
|
||||
if runErr != nil {
|
||||
if jsonOutput {
|
||||
_ = printJSON(map[string]any{"id": job.ID, "status": "failed", "error": runErr.Error()})
|
||||
@@ -206,7 +216,7 @@ func renderJobs(cmd *cobra.Command, jobs []*model.Job) error {
|
||||
return printJSON(jobs)
|
||||
}
|
||||
if len(jobs) == 0 {
|
||||
fmt.Fprintln(cmd.OutOrStdout(), "No jobs. Use 'orca job run <spec.hcl>' to submit one.")
|
||||
fmt.Fprintln(cmd.OutOrStdout(), "No jobs. Use 'orca job run <spec.md>' to submit one.")
|
||||
return nil
|
||||
}
|
||||
fmt.Fprintf(cmd.OutOrStdout(), "%-36s %-20s %-12s %-8s\n", "ID", "NAME", "STATUS", "EXIT")
|
||||
@@ -283,11 +293,79 @@ func renderJobTable(jobs []*model.Job) string {
|
||||
return out
|
||||
}
|
||||
|
||||
// jobStopTransport is the SSH command-execution seam used by
|
||||
// `orca job stop`. *sshpush.Transport satisfies it via Exec; tests
|
||||
// inject a mock (same pattern as driftTransport / drainExecer).
|
||||
type jobStopTransport interface {
|
||||
Exec(ctx context.Context, peer string, cmd string) ([]byte, error)
|
||||
}
|
||||
|
||||
// jobStopTransportOverride is the package-level test seam for the
|
||||
// SSH transport used by `orca job stop`. When non-nil it replaces the
|
||||
// production transport; tests set it and restore nil in cleanup.
|
||||
var jobStopTransportOverride jobStopTransport
|
||||
|
||||
// jobStopTimeout is the SSH command timeout for `orca job stop`.
|
||||
var jobStopTimeout time.Duration
|
||||
|
||||
// jobStopPeer is the optional --peer override for `orca job stop`.
|
||||
// When empty, the node is looked up from the alloc_history table
|
||||
// (latest entry for the job id). When set, the SSH stop targets that
|
||||
// peer directly.
|
||||
var jobStopPeer string
|
||||
|
||||
func jobStopTransportFromCtx() (jobStopTransport, error) {
|
||||
if jobStopTransportOverride != nil {
|
||||
return jobStopTransportOverride, nil
|
||||
}
|
||||
keyPath := certpaths.SSHKeyPath()
|
||||
khPath := certpaths.KnownHostsPath()
|
||||
return sshpush.NewTransport(keyPath, khPath), nil
|
||||
}
|
||||
|
||||
// nodeForJob looks up the node that ran (or is running) a job by
|
||||
// searching the alloc_history table for the latest entry for the
|
||||
// given job id. Returns nil if no history entry exists (the job may
|
||||
// have been run locally or pre-dates alloc_history).
|
||||
func nodeForJob(ctx context.Context, db *sql.DB, jobID string) (*model.Node, error) {
|
||||
hist := store.NewAllocHistoryRepo(db)
|
||||
if err := hist.EnsureSchema(ctx); err != nil {
|
||||
return nil, fmt.Errorf("alloc history schema: %w", err)
|
||||
}
|
||||
entries, err := hist.List(ctx, store.HistoryFilter{JobID: jobID})
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("alloc history list: %w", err)
|
||||
}
|
||||
if len(entries) == 0 {
|
||||
return nil, nil
|
||||
}
|
||||
// Pick the latest entry (List returns ASC; take the last).
|
||||
latest := entries[len(entries)-1]
|
||||
if latest.NodeID == "" {
|
||||
return nil, nil
|
||||
}
|
||||
nodeRepo := store.NewNodeRepo(db)
|
||||
n, err := nodeRepo.Get(ctx, latest.NodeID)
|
||||
if err != nil {
|
||||
if errors.Is(err, store.ErrNotFound) {
|
||||
return nil, nil
|
||||
}
|
||||
return nil, fmt.Errorf("lookup node %s: %w", latest.NodeID, err)
|
||||
}
|
||||
return n, nil
|
||||
}
|
||||
|
||||
var jobStopCmd = &cobra.Command{
|
||||
Use: "stop [job-id]",
|
||||
Short: "Stop a running job",
|
||||
Long: "Mark a job as stopped. Note: this is a soft stop (cancel context for the daemon).",
|
||||
Args: cobra.MaximumNArgs(1),
|
||||
Long: `Stop a running job by sending 'systemctl stop orca-alloc-<name>-*'
|
||||
to the node running the allocation via SSH, then mark the job as
|
||||
stopped in the DB (REQ-158, P09 T1).
|
||||
|
||||
If --peer is not given, the node is looked up from the allocation
|
||||
history. If no node is found, the DB status is updated anyway (soft
|
||||
stop fallback for local-run jobs).`,
|
||||
Args: cobra.MaximumNArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
id := stopID
|
||||
if id == "" && len(args) > 0 {
|
||||
@@ -296,8 +374,6 @@ var jobStopCmd = &cobra.Command{
|
||||
if id == "" {
|
||||
return fmt.Errorf("job id required (--id or argument)")
|
||||
}
|
||||
ctx, cancel := context.WithTimeout(cmd.Context(), 5*time.Second)
|
||||
defer cancel()
|
||||
|
||||
db, closer, err := openDB()
|
||||
if err != nil {
|
||||
@@ -305,6 +381,9 @@ var jobStopCmd = &cobra.Command{
|
||||
}
|
||||
defer closer()
|
||||
|
||||
ctx, cancel := context.WithTimeout(cmd.Context(), 30*time.Second)
|
||||
defer cancel()
|
||||
|
||||
repo := store.NewJobRepo(db)
|
||||
job, err := repo.Get(ctx, id)
|
||||
if err != nil {
|
||||
@@ -313,13 +392,73 @@ var jobStopCmd = &cobra.Command{
|
||||
}
|
||||
return err
|
||||
}
|
||||
|
||||
// Determine the peer to SSH to. --peer takes precedence;
|
||||
// otherwise look up the node from alloc_history.
|
||||
peer := jobStopPeer
|
||||
var node *model.Node
|
||||
if peer == "" {
|
||||
node, err = nodeForJob(ctx, db, id)
|
||||
if err != nil {
|
||||
return fmt.Errorf("lookup node for job %s: %w", id, err)
|
||||
}
|
||||
if node != nil {
|
||||
peer = peerAddrForNode(node)
|
||||
}
|
||||
}
|
||||
|
||||
// Validate the job name before interpolation into the shell
|
||||
// command (same injection guard as logs --job / stopAlloc).
|
||||
jobName := job.Name
|
||||
if !validSafeName(jobName) {
|
||||
return fmt.Errorf("job stop: invalid job name %q (allowed: A-Z a-z 0-9 _ -)", jobName)
|
||||
}
|
||||
|
||||
sshRan := false
|
||||
if peer != "" {
|
||||
transport, terr := jobStopTransportFromCtx()
|
||||
if terr != nil {
|
||||
return fmt.Errorf("job stop: ssh transport: %w", terr)
|
||||
}
|
||||
stopCtx, stopCancel := sshCmdCtx(ctx, jobStopTimeout)
|
||||
defer stopCancel()
|
||||
// Match the drift.go job restart unit pattern: orca-alloc-<name>.
|
||||
// Use a glob (orca-alloc-<name>-*) to stop all task units in
|
||||
// a multi-task allocation group.
|
||||
unitPattern := fmt.Sprintf("orca-alloc-%s-*", jobName)
|
||||
stopCmd := fmt.Sprintf("systemctl stop %s", shellQuote(unitPattern))
|
||||
out, sErr := transport.Exec(stopCtx, peer, stopCmd)
|
||||
if sErr != nil {
|
||||
// Non-fatal: the unit may not be running (already
|
||||
// stopped) or SSH may fail. We still update the DB
|
||||
// status so the operator's intent is recorded.
|
||||
fmt.Fprintf(cmd.ErrOrStderr(), "⚠ job stop: SSH systemctl stop failed on %s: %v (output: %s)\n", peer, sErr, strings.TrimSpace(string(out)))
|
||||
} else {
|
||||
sshRan = true
|
||||
}
|
||||
}
|
||||
|
||||
if err := repo.UpdateStatus(ctx, id, model.JobStatusStopped, 130); err != nil {
|
||||
return err
|
||||
}
|
||||
// REQ-156 / P07 T5: invalidate the jobs cache so the next
|
||||
// `orca job list` reflects the just-stopped job.
|
||||
cacheInvalidate(cacheJobClass)
|
||||
if jsonOutput {
|
||||
return printJSON(map[string]any{"id": id, "status": "stopped", "previous_status": job.Status})
|
||||
result := map[string]any{"id": id, "status": "stopped", "previous_status": job.Status}
|
||||
if peer != "" {
|
||||
result["peer"] = peer
|
||||
result["ssh_stop"] = sshRan
|
||||
}
|
||||
return printJSON(result)
|
||||
}
|
||||
if peer != "" && sshRan {
|
||||
fmt.Fprintf(cmd.OutOrStdout(), "✓ Job stopped: %s (systemctl stop on %s)\n", id, peer)
|
||||
} else if peer != "" {
|
||||
fmt.Fprintf(cmd.OutOrStdout(), "✓ Job stopped: %s (DB only; SSH stop failed — see stderr)\n", id)
|
||||
} else {
|
||||
fmt.Fprintf(cmd.OutOrStdout(), "✓ Job stopped: %s (DB only; no node found)\n", id)
|
||||
}
|
||||
fmt.Fprintf(cmd.OutOrStdout(), "✓ Job stopped: %s\n", id)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
@@ -373,6 +512,8 @@ var jobLogsCmd = &cobra.Command{
|
||||
|
||||
func init() {
|
||||
jobStopCmd.Flags().StringVar(&stopID, "id", "", "job id")
|
||||
jobStopCmd.Flags().StringVar(&jobStopPeer, "peer", "", "peer address (host:port) running the allocation (auto-detected from alloc history if empty)")
|
||||
jobStopCmd.Flags().DurationVar(&jobStopTimeout, "timeout", sshCmdDefaultTimeout, "SSH command timeout")
|
||||
jobLogsCmd.Flags().StringVar(&stopID, "id", "", "job id")
|
||||
jobRunCmd.Flags().StringVar(&runTarget, "target", "", "pin job to a specific node id (overrides bin-packing)")
|
||||
jobRunCmd.Flags().StringVar(&runIDKey, "idempotency-key", "", "X-Orca-Idempotency-Key for cross-node dispatch dedupe")
|
||||
|
||||
@@ -2,11 +2,14 @@ package cli
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"encoding/json"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"git.cloudinit.dev/coreci/orca/internal/certpaths"
|
||||
"git.cloudinit.dev/coreci/orca/internal/model"
|
||||
@@ -310,3 +313,193 @@ func seedJob(t *testing.T, name string, status model.JobStatus) string {
|
||||
}
|
||||
return j.ID
|
||||
}
|
||||
|
||||
|
||||
// mockJobStopExec is a record-and-replay SSH execer for `orca job stop`
|
||||
// tests (same pattern as mockDrainExec / mockLogsExec).
|
||||
type mockJobStopExec struct {
|
||||
mu sync.Mutex
|
||||
responses []jobStopMockResp
|
||||
calls []jobStopMockCall
|
||||
}
|
||||
|
||||
type jobStopMockResp struct {
|
||||
match string
|
||||
out string
|
||||
exit int
|
||||
}
|
||||
|
||||
type jobStopMockCall struct {
|
||||
peer string
|
||||
cmd string
|
||||
}
|
||||
|
||||
func (m *mockJobStopExec) Exec(_ context.Context, peer, cmd string) ([]byte, error) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
m.calls = append(m.calls, jobStopMockCall{peer: peer, cmd: cmd})
|
||||
for _, r := range m.responses {
|
||||
if r.match == "" || strings.Contains(cmd, r.match) {
|
||||
return []byte(r.out), nil
|
||||
}
|
||||
}
|
||||
return []byte(""), nil
|
||||
}
|
||||
|
||||
func (m *mockJobStopExec) callsFor(match string) []jobStopMockCall {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
var out []jobStopMockCall
|
||||
for _, c := range m.calls {
|
||||
if strings.Contains(c.cmd, match) {
|
||||
out = append(out, c)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// TestJobStopSSH verifies that `orca job stop` sends a real
|
||||
// 'systemctl stop' via SSH to the target node when the job has a
|
||||
// recorded allocation history (REQ-158, P09 T6).
|
||||
func TestJobStopSSH(t *testing.T) {
|
||||
_, cleanup := initTestEnv(t)
|
||||
defer cleanup()
|
||||
if err := runInit(discardWriter{}); err != nil {
|
||||
t.Fatalf("init: %v", err)
|
||||
}
|
||||
|
||||
// Seed a node and a job, then record an alloc_history entry
|
||||
// linking the job to the node.
|
||||
nodeID := seedNode(t, "worker-1", "worker-1:8443")
|
||||
jobID := seedJob(t, "webapp", model.JobStatusRunning)
|
||||
|
||||
db, err := store.Open(certpaths.DBPath())
|
||||
if err != nil {
|
||||
t.Fatalf("open db: %v", err)
|
||||
}
|
||||
defer db.Close()
|
||||
hist := store.NewAllocHistoryRepo(db)
|
||||
ctx := context.Background()
|
||||
if err := hist.EnsureSchema(ctx); err != nil {
|
||||
t.Fatalf("ensure schema: %v", err)
|
||||
}
|
||||
if err := hist.Record(ctx, store.AllocHistoryEntry{
|
||||
AllocID: "default/webapp-0",
|
||||
JobID: jobID,
|
||||
NodeID: nodeID,
|
||||
Namespace: "default",
|
||||
ToState: "created",
|
||||
Timestamp: time.Now().UTC(),
|
||||
}); err != nil {
|
||||
t.Fatalf("record alloc history: %v", err)
|
||||
}
|
||||
|
||||
// Wire the mock SSH transport (must be after resetRootFlags so
|
||||
// resetCommandFlags doesn't nil it out).
|
||||
resetRootFlags(t)
|
||||
mock := &mockJobStopExec{}
|
||||
prev := jobStopTransportOverride
|
||||
jobStopTransportOverride = mock
|
||||
t.Cleanup(func() { jobStopTransportOverride = prev })
|
||||
|
||||
var buf bytes.Buffer
|
||||
rootCmd.SetOut(&buf)
|
||||
rootCmd.SetErr(&buf)
|
||||
rootCmd.SetArgs([]string{"job", "stop", jobID})
|
||||
if err := rootCmd.Execute(); err != nil {
|
||||
t.Fatalf("job stop: %v", err)
|
||||
}
|
||||
|
||||
// Verify systemctl stop was called via SSH.
|
||||
stopCalls := mock.callsFor("systemctl stop")
|
||||
if len(stopCalls) == 0 {
|
||||
t.Fatalf("expected systemctl stop SSH call, got %d calls: %v", len(mock.calls), mock.calls)
|
||||
}
|
||||
if !strings.Contains(stopCalls[0].cmd, "orca-alloc-webapp-*") {
|
||||
t.Errorf("expected 'orca-alloc-webapp-*' in cmd, got: %s", stopCalls[0].cmd)
|
||||
}
|
||||
if !strings.Contains(stopCalls[0].peer, "worker-1") {
|
||||
t.Errorf("expected peer to contain 'worker-1', got: %s", stopCalls[0].peer)
|
||||
}
|
||||
|
||||
// Verify the DB status was updated.
|
||||
repo := store.NewJobRepo(db)
|
||||
job, err := repo.Get(ctx, jobID)
|
||||
if err != nil {
|
||||
t.Fatalf("get job: %v", err)
|
||||
}
|
||||
if job.Status != model.JobStatusStopped {
|
||||
t.Errorf("job status = %v, want stopped", job.Status)
|
||||
}
|
||||
}
|
||||
|
||||
// TestJobStopSSHPeerOverride verifies that --peer bypasses the
|
||||
// alloc_history lookup and uses the given peer directly (REQ-158).
|
||||
func TestJobStopSSHPeerOverride(t *testing.T) {
|
||||
_, cleanup := initTestEnv(t)
|
||||
defer cleanup()
|
||||
if err := runInit(discardWriter{}); err != nil {
|
||||
t.Fatalf("init: %v", err)
|
||||
}
|
||||
|
||||
jobID := seedJob(t, "webapp2", model.JobStatusRunning)
|
||||
|
||||
resetRootFlags(t)
|
||||
mock := &mockJobStopExec{}
|
||||
prev := jobStopTransportOverride
|
||||
jobStopTransportOverride = mock
|
||||
t.Cleanup(func() { jobStopTransportOverride = prev })
|
||||
|
||||
var buf bytes.Buffer
|
||||
rootCmd.SetOut(&buf)
|
||||
rootCmd.SetErr(&buf)
|
||||
rootCmd.SetArgs([]string{"job", "stop", jobID, "--peer", "10.0.0.5:22"})
|
||||
if err := rootCmd.Execute(); err != nil {
|
||||
t.Fatalf("job stop: %v", err)
|
||||
}
|
||||
|
||||
stopCalls := mock.callsFor("systemctl stop")
|
||||
if len(stopCalls) == 0 {
|
||||
t.Fatalf("expected systemctl stop SSH call, got %d calls", len(mock.calls))
|
||||
}
|
||||
if stopCalls[0].peer != "10.0.0.5:22" {
|
||||
t.Errorf("peer = %s, want 10.0.0.5:22", stopCalls[0].peer)
|
||||
}
|
||||
}
|
||||
|
||||
// TestJobStopNoNodeFallback verifies that when no node is found in
|
||||
// alloc_history, the job is still stopped in the DB (soft stop
|
||||
// fallback) without attempting SSH (REQ-158).
|
||||
func TestJobStopNoNodeFallback(t *testing.T) {
|
||||
_, cleanup := initTestEnv(t)
|
||||
defer cleanup()
|
||||
if err := runInit(discardWriter{}); err != nil {
|
||||
t.Fatalf("init: %v", err)
|
||||
}
|
||||
|
||||
jobID := seedJob(t, "localjob", model.JobStatusRunning)
|
||||
|
||||
resetRootFlags(t)
|
||||
mock := &mockJobStopExec{}
|
||||
prev := jobStopTransportOverride
|
||||
jobStopTransportOverride = mock
|
||||
t.Cleanup(func() { jobStopTransportOverride = prev })
|
||||
|
||||
var buf bytes.Buffer
|
||||
rootCmd.SetOut(&buf)
|
||||
rootCmd.SetErr(&buf)
|
||||
rootCmd.SetArgs([]string{"job", "stop", jobID})
|
||||
if err := rootCmd.Execute(); err != nil {
|
||||
t.Fatalf("job stop: %v", err)
|
||||
}
|
||||
|
||||
// No SSH calls should have been made (no node found).
|
||||
if len(mock.calls) > 0 {
|
||||
t.Errorf("expected 0 SSH calls, got %d: %v", len(mock.calls), mock.calls)
|
||||
}
|
||||
|
||||
// Verify the output mentions "DB only".
|
||||
if !strings.Contains(buf.String(), "DB only") {
|
||||
t.Errorf("output should mention 'DB only', got: %s", buf.String())
|
||||
}
|
||||
}
|
||||
|
||||
+43
-6
@@ -101,8 +101,20 @@ var (
|
||||
logsJob string
|
||||
logsSince string
|
||||
logsJSON bool
|
||||
logsLines int
|
||||
)
|
||||
|
||||
// logsMaxLines is the hard cap on --lines to prevent OOM from
|
||||
// unbounded journalctl output (REQ-158, P09 T3).
|
||||
const logsMaxLines = 50000
|
||||
|
||||
// logsDefaultLines is the default --lines value.
|
||||
const logsDefaultLines = 1000
|
||||
|
||||
// logsMaxSince is the maximum lookback for --since (7 days) to
|
||||
// prevent OOM from unbounded journalctl queries (REQ-158, P09 T3).
|
||||
const logsMaxSince = 7 * 24 * time.Hour
|
||||
|
||||
var logsCmd = &cobra.Command{
|
||||
Use: "logs",
|
||||
Short: "Aggregate journald logs across nodes (REQ-117)",
|
||||
@@ -139,6 +151,25 @@ Ctrl-C cancels the fan-out via signal.NotifyContext.`,
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
// REQ-158 / P09 T3: clamp --since to 7 days max to prevent
|
||||
// OOM from unbounded journalctl queries. If the requested
|
||||
// lookback exceeds the cap, clamp it and warn.
|
||||
now := time.Now().UTC()
|
||||
maxSince := now.Add(-logsMaxSince)
|
||||
if since.Before(maxSince) {
|
||||
fmt.Fprintf(cmd.ErrOrStderr(), "⚠ --since %s exceeds 7d cap; clamping to 7d\n", logsSince)
|
||||
since = maxSince
|
||||
}
|
||||
|
||||
// REQ-158 / P09 T3: clamp --lines to [1, logsMaxLines].
|
||||
lines := logsLines
|
||||
if lines <= 0 {
|
||||
lines = logsDefaultLines
|
||||
}
|
||||
if lines > logsMaxLines {
|
||||
fmt.Fprintf(cmd.ErrOrStderr(), "⚠ --lines %d exceeds max %d; clamping\n", lines, logsMaxLines)
|
||||
lines = logsMaxLines
|
||||
}
|
||||
|
||||
ctx, cancel := signal.NotifyContext(cmd.Context(), os.Interrupt, syscall.SIGTERM)
|
||||
defer cancel()
|
||||
@@ -158,7 +189,7 @@ Ctrl-C cancels the fan-out via signal.NotifyContext.`,
|
||||
|
||||
out := cmd.OutOrStdout()
|
||||
multi := len(nodes) > 1
|
||||
for line := range streamLogs(ctx, ex, nodes, since, logsJob) {
|
||||
for line := range streamLogs(ctx, ex, nodes, since, logsJob, lines) {
|
||||
if logsJSON {
|
||||
raw, _ := json.Marshal(line)
|
||||
fmt.Fprintln(out, string(raw))
|
||||
@@ -227,7 +258,7 @@ func resolveLogNodes(ctx context.Context) ([]*model.Node, error) {
|
||||
// JSON entry immediately. The stream ends when every node has
|
||||
// completed (or the context is cancelled). The caller drives the
|
||||
// iteration via range-over-func (D-017 iter.Seq pattern).
|
||||
func streamLogs(ctx context.Context, ex logsExecer, nodes []*model.Node, since time.Time, job string) iter.Seq[LogLine] {
|
||||
func streamLogs(ctx context.Context, ex logsExecer, nodes []*model.Node, since time.Time, job string, lines int) iter.Seq[LogLine] {
|
||||
return func(yield func(LogLine) bool) {
|
||||
merged := make(chan LogLine)
|
||||
var wg sync.WaitGroup
|
||||
@@ -235,7 +266,7 @@ func streamLogs(ctx context.Context, ex logsExecer, nodes []*model.Node, since t
|
||||
wg.Add(1)
|
||||
go func(n *model.Node) {
|
||||
defer wg.Done()
|
||||
streamNodeLines(ctx, ex, n, since, job, merged)
|
||||
streamNodeLines(ctx, ex, n, since, job, lines, merged)
|
||||
}(n)
|
||||
}
|
||||
done := make(chan struct{})
|
||||
@@ -267,7 +298,7 @@ func streamLogs(ctx context.Context, ex logsExecer, nodes []*model.Node, since t
|
||||
// context is cancelled); the caller is responsible for waiting on the
|
||||
// goroutine. Send is non-blocking via select on ctx.Done so a slow
|
||||
// consumer does not stall the fanout forever.
|
||||
func streamNodeLines(ctx context.Context, ex logsExecer, n *model.Node, since time.Time, job string, out chan<- LogLine) {
|
||||
func streamNodeLines(ctx context.Context, ex logsExecer, n *model.Node, since time.Time, job string, lines int, out chan<- LogLine) {
|
||||
peer := peerAddrForNode(n)
|
||||
if peer == "" {
|
||||
slog.Default().Warn("logs: cannot resolve SSH address for node", "node", n.Name)
|
||||
@@ -278,9 +309,14 @@ func streamNodeLines(ctx context.Context, ex logsExecer, n *model.Node, since ti
|
||||
unitPattern = "orca-alloc-" + job + "-*"
|
||||
}
|
||||
sinceStr := since.Format("2006-01-02 15:04:05")
|
||||
// REQ-158 / P09 T3: pass --lines=N to journalctl to cap output
|
||||
// and prevent OOM from unbounded log queries.
|
||||
if lines <= 0 {
|
||||
lines = logsDefaultLines
|
||||
}
|
||||
// F1: shellQuote (single-quote wrap) instead of %q — %q does not
|
||||
// escape backticks, enabling command substitution in double quotes.
|
||||
cmd := fmt.Sprintf("journalctl -u %s --since %s --output json --no-pager", shellQuote(unitPattern), shellQuote(sinceStr))
|
||||
cmd := fmt.Sprintf("journalctl -u %s --since %s --lines %d --output json --no-pager", shellQuote(unitPattern), shellQuote(sinceStr), lines)
|
||||
raw, err := ex.Exec(ctx, peer, cmd)
|
||||
if err != nil {
|
||||
slog.Default().Warn("logs: exec failed", "node", n.Name, "peer", peer, "error", err)
|
||||
@@ -324,7 +360,8 @@ func init() {
|
||||
logsCmd.Flags().BoolVar(&logsAllNodes, "all-nodes", false, "fan out to all registered nodes")
|
||||
logsCmd.Flags().StringVar(&logsNode, "node", "", "restrict to a single node (name or id)")
|
||||
logsCmd.Flags().StringVar(&logsJob, "job", "", "filter by job name (matches orca-alloc-<name>-* units)")
|
||||
logsCmd.Flags().StringVar(&logsSince, "since", "5m", "duration lookback (e.g. 5m, 1h, 30m); default 5m")
|
||||
logsCmd.Flags().StringVar(&logsSince, "since", "5m", "duration lookback (e.g. 5m, 1h, 30m); default 5m; max 7d")
|
||||
logsCmd.Flags().IntVar(&logsLines, "lines", logsDefaultLines, fmt.Sprintf("max number of journal lines per node (default %d, max %d)", logsDefaultLines, logsMaxLines))
|
||||
logsCmd.Flags().BoolVar(&logsJSON, "json", false, "output raw JSON (one LogLine per line)")
|
||||
rootCmd.AddCommand(logsCmd)
|
||||
}
|
||||
|
||||
+156
-1
@@ -63,6 +63,18 @@ func (m *mockLogsExec) countCalls(match string) int {
|
||||
return n
|
||||
}
|
||||
|
||||
func (m *mockLogsExec) callsFor(match string) []logsMockCall {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
var out []logsMockCall
|
||||
for _, c := range m.calls {
|
||||
if strings.Contains(c.cmd, match) {
|
||||
out = append(out, c)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// logsTestEnv wires a mockLogsExec into logsExecOverride and returns
|
||||
// the mock + a cleanup func. Tests MUST defer the cleanup.
|
||||
func logsTestEnv(t *testing.T) *mockLogsExec {
|
||||
@@ -325,7 +337,7 @@ func TestLogsCancelStopsStream(t *testing.T) {
|
||||
{ID: "n1", Name: "cancelnode", Address: "cancelnode:8443"},
|
||||
}
|
||||
consumed := 0
|
||||
for range streamLogs(ctx, ex, nodes, time.Now().UTC().Add(-1*time.Minute), "") {
|
||||
for range streamLogs(ctx, ex, nodes, time.Now().UTC().Add(-1*time.Minute), "", 1000) {
|
||||
consumed++
|
||||
}
|
||||
if consumed > 1 {
|
||||
@@ -357,3 +369,146 @@ func TestLogsParseJournalLine_InvalidJSON(t *testing.T) {
|
||||
t.Error("expected error for invalid json, got nil")
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
// TestLogsLinesFlag verifies that --lines is passed through to the
|
||||
// journalctl command as --lines=N (REQ-158, P09 T8).
|
||||
func TestLogsLinesFlag(t *testing.T) {
|
||||
_, cleanup := initTestEnv(t)
|
||||
defer cleanup()
|
||||
|
||||
logsNodeForTest(t, "linesnode", "linesnode:8443")
|
||||
mx := logsTestEnv(t)
|
||||
|
||||
ts := time.Date(2026, 1, 1, 12, 0, 0, 0, time.UTC)
|
||||
mx.responses = []logsMockResp{
|
||||
{match: "journalctl", out: journalJSONLine(ts, "orca-alloc-web-0", "line test", "6") + "\n"},
|
||||
}
|
||||
|
||||
_, err := runLogsCmd(t, []string{"logs", "--node", "linesnode", "--since", "1m", "--lines", "500"})
|
||||
if err != nil {
|
||||
t.Fatalf("logs: %v", err)
|
||||
}
|
||||
calls := mx.callsFor("journalctl")
|
||||
if len(calls) == 0 {
|
||||
t.Fatal("expected journalctl call")
|
||||
}
|
||||
if !strings.Contains(calls[0].cmd, "--lines 500") {
|
||||
t.Errorf("expected '--lines 500' in cmd, got: %s", calls[0].cmd)
|
||||
}
|
||||
}
|
||||
|
||||
// TestLogsLinesDefault verifies that the default --lines value (1000)
|
||||
// is passed to journalctl when --lines is not specified (REQ-158).
|
||||
func TestLogsLinesDefault(t *testing.T) {
|
||||
_, cleanup := initTestEnv(t)
|
||||
defer cleanup()
|
||||
|
||||
logsNodeForTest(t, "defnode", "defnode:8443")
|
||||
mx := logsTestEnv(t)
|
||||
|
||||
ts := time.Date(2026, 1, 1, 12, 0, 0, 0, time.UTC)
|
||||
mx.responses = []logsMockResp{
|
||||
{match: "journalctl", out: journalJSONLine(ts, "orca-alloc-web-0", "default lines", "6") + "\n"},
|
||||
}
|
||||
|
||||
_, err := runLogsCmd(t, []string{"logs", "--node", "defnode", "--since", "1m"})
|
||||
if err != nil {
|
||||
t.Fatalf("logs: %v", err)
|
||||
}
|
||||
calls := mx.callsFor("journalctl")
|
||||
if len(calls) == 0 {
|
||||
t.Fatal("expected journalctl call")
|
||||
}
|
||||
if !strings.Contains(calls[0].cmd, "--lines 1000") {
|
||||
t.Errorf("expected default '--lines 1000' in cmd, got: %s", calls[0].cmd)
|
||||
}
|
||||
}
|
||||
|
||||
// TestLogsLinesClamp verifies that --lines exceeding the max (50000) is
|
||||
// clamped (REQ-158, P09 T8).
|
||||
func TestLogsLinesClamp(t *testing.T) {
|
||||
_, cleanup := initTestEnv(t)
|
||||
defer cleanup()
|
||||
|
||||
logsNodeForTest(t, "clampnode", "clampnode:8443")
|
||||
mx := logsTestEnv(t)
|
||||
|
||||
ts := time.Date(2026, 1, 1, 12, 0, 0, 0, time.UTC)
|
||||
mx.responses = []logsMockResp{
|
||||
{match: "journalctl", out: journalJSONLine(ts, "orca-alloc-web-0", "clamp test", "6") + "\n"},
|
||||
}
|
||||
|
||||
_, err := runLogsCmd(t, []string{"logs", "--node", "clampnode", "--since", "1m", "--lines", "999999"})
|
||||
if err != nil {
|
||||
t.Fatalf("logs: %v", err)
|
||||
}
|
||||
calls := mx.callsFor("journalctl")
|
||||
if len(calls) == 0 {
|
||||
t.Fatal("expected journalctl call")
|
||||
}
|
||||
if !strings.Contains(calls[0].cmd, "--lines 50000") {
|
||||
t.Errorf("expected clamped '--lines 50000' in cmd, got: %s", calls[0].cmd)
|
||||
}
|
||||
}
|
||||
|
||||
// TestLogsSinceClamp verifies that --since exceeding 7 days is
|
||||
// clamped and a warning is printed (REQ-158, P09 T8).
|
||||
func TestLogsSinceClamp(t *testing.T) {
|
||||
_, cleanup := initTestEnv(t)
|
||||
defer cleanup()
|
||||
|
||||
logsNodeForTest(t, "sincenode", "sincenode:8443")
|
||||
mx := logsTestEnv(t)
|
||||
|
||||
ts := time.Date(2026, 1, 1, 12, 0, 0, 0, time.UTC)
|
||||
mx.responses = []logsMockResp{
|
||||
{match: "journalctl", out: journalJSONLine(ts, "orca-alloc-web-0", "since test", "6") + "\n"},
|
||||
}
|
||||
|
||||
var buf bytes.Buffer
|
||||
rootCmd.SetOut(&buf)
|
||||
rootCmd.SetErr(&buf)
|
||||
resetRootFlags(t)
|
||||
rootCmd.SetOut(&buf)
|
||||
rootCmd.SetErr(&buf)
|
||||
rootCmd.SetArgs([]string{"logs", "--node", "sincenode", "--since", "720h"})
|
||||
err := rootCmd.Execute()
|
||||
if err != nil {
|
||||
t.Fatalf("logs: %v", err)
|
||||
}
|
||||
out := buf.String()
|
||||
if !strings.Contains(out, "clamping to 7d") {
|
||||
t.Errorf("expected warning about clamping --since to 7d, got: %s", out)
|
||||
}
|
||||
calls := mx.callsFor("journalctl")
|
||||
if len(calls) == 0 {
|
||||
t.Fatal("expected journalctl call")
|
||||
}
|
||||
}
|
||||
|
||||
// TestLogsLinesFlagJSON verifies --lines is passed through in JSON mode.
|
||||
func TestLogsLinesFlagJSON(t *testing.T) {
|
||||
_, cleanup := initTestEnv(t)
|
||||
defer cleanup()
|
||||
|
||||
logsNodeForTest(t, "jsonlines", "jsonlines:8443")
|
||||
mx := logsTestEnv(t)
|
||||
|
||||
ts := time.Date(2026, 1, 1, 12, 0, 0, 0, time.UTC)
|
||||
mx.responses = []logsMockResp{
|
||||
{match: "journalctl", out: journalJSONLine(ts, "orca-alloc-web-0", "json lines test", "6") + "\n"},
|
||||
}
|
||||
|
||||
_, err := runLogsCmd(t, []string{"logs", "--node", "jsonlines", "--since", "1m", "--lines", "200", "--json"})
|
||||
if err != nil {
|
||||
t.Fatalf("logs: %v", err)
|
||||
}
|
||||
calls := mx.callsFor("journalctl")
|
||||
if len(calls) == 0 {
|
||||
t.Fatal("expected journalctl call")
|
||||
}
|
||||
if !strings.Contains(calls[0].cmd, "--lines 200") {
|
||||
t.Errorf("expected '--lines 200' in cmd, got: %s", calls[0].cmd)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -14,6 +14,7 @@ import (
|
||||
|
||||
"github.com/spf13/cobra"
|
||||
|
||||
"git.cloudinit.dev/coreci/orca/internal/model"
|
||||
"git.cloudinit.dev/coreci/orca/internal/store"
|
||||
"git.cloudinit.dev/coreci/orca/internal/transport"
|
||||
)
|
||||
@@ -54,12 +55,16 @@ updates gauges. No orca daemon required (R-001).`,
|
||||
|
||||
mux := http.NewServeMux()
|
||||
mux.HandleFunc("/metrics", func(w http.ResponseWriter, r *http.Request) {
|
||||
w.Header().Set("X-Content-Type-Options", "nosniff")
|
||||
w.Header().Set("X-Frame-Options", "DENY")
|
||||
w.Header().Set("Content-Type", "text/plain; version=0.0.4; charset=utf-8")
|
||||
if err := m.WritePrometheus(w); err != nil {
|
||||
log.Warn("metrics: write exposition failed", "err", err)
|
||||
}
|
||||
})
|
||||
mux.HandleFunc("/healthz", func(w http.ResponseWriter, r *http.Request) {
|
||||
w.Header().Set("X-Content-Type-Options", "nosniff")
|
||||
w.Header().Set("X-Frame-Options", "DENY")
|
||||
w.WriteHeader(http.StatusOK)
|
||||
_, _ = w.Write([]byte("ok\n"))
|
||||
})
|
||||
@@ -124,6 +129,22 @@ func refresh(ctx context.Context, m *transport.Metrics, db *sql.DB, log interfac
|
||||
log.Warn("metrics: job list failed", "err", err)
|
||||
} else {
|
||||
m.SetGauge("allocs_total", float64(len(jobs)))
|
||||
// REQ-159 / P10: jobs by state.
|
||||
byState := make(map[model.JobStatus]int, 8)
|
||||
for _, j := range jobs {
|
||||
byState[j.Status]++
|
||||
}
|
||||
// Set total + per-state counts using simple gauge names.
|
||||
running := byState[model.JobStatusRunning]
|
||||
failed := byState[model.JobStatusFailed]
|
||||
complete := byState[model.JobStatusComplete]
|
||||
m.SetGauge("orca_jobs_running", float64(running))
|
||||
m.SetGauge("orca_jobs_failed", float64(failed))
|
||||
m.SetGauge("orca_jobs_complete", float64(complete))
|
||||
}
|
||||
// REQ-159 / P10: audit chain head gauge.
|
||||
if head, err := store.NewAuditRepo(db).ChainHead(ctx); err == nil && head != "" {
|
||||
m.SetGauge("orca_audit_chain_head", 1)
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+60
-102
@@ -1,116 +1,74 @@
|
||||
package cli
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"io"
|
||||
"net"
|
||||
"net/http"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"git.cloudinit.dev/coreci/orca/internal/model"
|
||||
"git.cloudinit.dev/coreci/orca/internal/store"
|
||||
"git.cloudinit.dev/coreci/orca/internal/transport"
|
||||
)
|
||||
|
||||
func TestMetricsCmdRegistered(t *testing.T) {
|
||||
found := false
|
||||
for _, c := range rootCmd.Commands() {
|
||||
if c.Name() == "metrics" {
|
||||
found = true
|
||||
break
|
||||
}
|
||||
// TestREQ159_MetricsExpanded verifies the expanded metric set (P10, REQ-159).
|
||||
func TestREQ159_MetricsExpanded(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
t.Setenv("ORCA_HOME", dir)
|
||||
dbPath := filepath.Join(dir, "orca.db")
|
||||
db, err := store.Open(dbPath)
|
||||
if err != nil {
|
||||
t.Fatalf("store.Open: %v", err)
|
||||
}
|
||||
if !found {
|
||||
t.Fatal("metricsCmd not registered on root")
|
||||
defer db.Close()
|
||||
|
||||
// Seed a node.
|
||||
repo := store.NewNodeRepo(db)
|
||||
if err := repo.Insert(context.Background(), &model.Node{
|
||||
ID: "test-node-1",
|
||||
Name: "test-node",
|
||||
Address: "localhost:8443",
|
||||
Kind: "localhost",
|
||||
OS: "linux",
|
||||
State: "ready",
|
||||
}); err != nil {
|
||||
t.Fatalf("insert node: %v", err)
|
||||
}
|
||||
|
||||
// Seed a job.
|
||||
jobRepo := store.NewJobRepo(db)
|
||||
if err := jobRepo.Insert(context.Background(), &model.Job{
|
||||
ID: "job-1",
|
||||
Name: "test-job",
|
||||
Status: "running",
|
||||
}); err != nil {
|
||||
t.Fatalf("insert job: %v", err)
|
||||
}
|
||||
|
||||
m := transport.NewMetrics()
|
||||
logger := slogLogger{}
|
||||
refresh(context.Background(), m, db, logger)
|
||||
|
||||
// Verify expanded metrics by reading the exposition output.
|
||||
var buf strings.Builder
|
||||
if err := m.WritePrometheus(&buf); err != nil {
|
||||
t.Fatalf("WritePrometheus: %v", err)
|
||||
}
|
||||
out := buf.String()
|
||||
if !strings.Contains(out, "nodes_total 1") {
|
||||
t.Errorf("output missing nodes_total 1:\n%s", out)
|
||||
}
|
||||
if !strings.Contains(out, "allocs_total 1") {
|
||||
t.Errorf("output missing allocs_total 1:\n%s", out)
|
||||
}
|
||||
if !strings.Contains(out, "orca_jobs_running") {
|
||||
t.Errorf("output missing orca_jobs_running:\n%s", out)
|
||||
}
|
||||
}
|
||||
|
||||
func TestMetricsAddrFlagDefault(t *testing.T) {
|
||||
f := metricsCmd.Flags().Lookup("addr")
|
||||
if f == nil {
|
||||
t.Fatal("--addr flag not registered on metricsCmd")
|
||||
}
|
||||
if f.DefValue != ":9100" {
|
||||
t.Errorf("--addr default = %q, want %q", f.DefValue, ":9100")
|
||||
}
|
||||
}
|
||||
type slogLogger struct{}
|
||||
|
||||
func TestMetricsEndpoints(t *testing.T) {
|
||||
_, cleanup := initTestEnv(t)
|
||||
defer cleanup()
|
||||
resetRootFlags(t)
|
||||
func (slogLogger) Warn(msg string, args ...any) {}
|
||||
|
||||
// Pick a free port by briefly listening then closing.
|
||||
ln, err := net.Listen("tcp", "127.0.0.1:0")
|
||||
if err != nil {
|
||||
t.Fatalf("listen probe: %v", err)
|
||||
}
|
||||
addr := ln.Addr().String()
|
||||
_ = ln.Close()
|
||||
|
||||
metricsAddr = addr
|
||||
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
defer cancel()
|
||||
|
||||
cmd := metricsCmd
|
||||
var out bytes.Buffer
|
||||
cmd.SetOut(&out)
|
||||
cmd.SetErr(&out)
|
||||
cmd.SetContext(ctx)
|
||||
|
||||
errCh := make(chan error, 1)
|
||||
go func() {
|
||||
errCh <- cmd.RunE(cmd, nil)
|
||||
}()
|
||||
|
||||
deadline := time.Now().Add(5 * time.Second)
|
||||
var resp *http.Response
|
||||
for time.Now().Before(deadline) {
|
||||
resp, err = http.Get("http://" + addr + "/healthz")
|
||||
if err == nil {
|
||||
break
|
||||
}
|
||||
time.Sleep(20 * time.Millisecond)
|
||||
}
|
||||
if err != nil {
|
||||
t.Fatalf("GET /healthz: %v", err)
|
||||
}
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Errorf("/healthz status = %d, want 200", resp.StatusCode)
|
||||
}
|
||||
body, _ := io.ReadAll(resp.Body)
|
||||
resp.Body.Close()
|
||||
if !strings.HasPrefix(string(body), "ok") {
|
||||
t.Errorf("/healthz body = %q, want \"ok\"", string(body))
|
||||
}
|
||||
|
||||
resp2, err := http.Get("http://" + addr + "/metrics")
|
||||
if err != nil {
|
||||
t.Fatalf("GET /metrics: %v", err)
|
||||
}
|
||||
defer resp2.Body.Close()
|
||||
if resp2.StatusCode != http.StatusOK {
|
||||
t.Errorf("/metrics status = %d, want 200", resp2.StatusCode)
|
||||
}
|
||||
mbody, _ := io.ReadAll(resp2.Body)
|
||||
ms := string(mbody)
|
||||
for _, name := range []string{
|
||||
"txns_applied_total",
|
||||
"txns_drifted_total",
|
||||
"drifts_remediated_total",
|
||||
"peers_total",
|
||||
"nodes_total",
|
||||
"allocs_total",
|
||||
} {
|
||||
if !strings.Contains(ms, name) {
|
||||
t.Errorf("/metrics missing %q\n---\n%s", name, ms)
|
||||
}
|
||||
}
|
||||
|
||||
cancel()
|
||||
select {
|
||||
case <-errCh:
|
||||
case <-time.After(3 * time.Second):
|
||||
t.Fatal("metrics command did not stop after cancel")
|
||||
}
|
||||
}
|
||||
var _ = os.Stdin
|
||||
|
||||
@@ -57,6 +57,11 @@ func resetCommandFlags() {
|
||||
driftConfigPath = ""
|
||||
driftRemediateForce = false
|
||||
jobRestartPeer = ""
|
||||
jobStopPeer = ""
|
||||
jobStopTimeout = 0
|
||||
jobStopTransportOverride = nil
|
||||
logsLines = logsDefaultLines
|
||||
cutoverFSOverride = nil
|
||||
jobLintExplain = false
|
||||
jobLintFormat = "text"
|
||||
jobVerifyLead = ""
|
||||
|
||||
+82
-2
@@ -15,6 +15,7 @@ import (
|
||||
"github.com/spf13/cobra"
|
||||
|
||||
"git.cloudinit.dev/coreci/orca/internal/certpaths"
|
||||
"git.cloudinit.dev/coreci/orca/internal/linux"
|
||||
"git.cloudinit.dev/coreci/orca/internal/engine"
|
||||
"git.cloudinit.dev/coreci/orca/internal/model"
|
||||
"git.cloudinit.dev/coreci/orca/internal/proxmox"
|
||||
@@ -73,16 +74,22 @@ var nodeJoinCmd = &cobra.Command{
|
||||
|
||||
Node types (via --type):
|
||||
localhost (default): register a local or Linux node (existing behavior)
|
||||
linux: SSH-bootstrap a remote generic Linux worker
|
||||
(Ubuntu/Debian/Alpine; deploys orca pubkey, creates orca
|
||||
user + drift-events dir; requires --host + --ssh-key)
|
||||
proxmox: SSH-bootstrap a remote Proxmox VE 8/9 host
|
||||
(deploys orca pubkey, creates orca user + PVE role +
|
||||
sudoers allowlist; requires --host + --ssh-key (R-021: no passwords))`,
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
if joinHostKeyFP != "" && joinType != "proxmox" {
|
||||
return fmt.Errorf("--host-key-fingerprint requires --type proxmox today")
|
||||
if joinHostKeyFP != "" && joinType != "proxmox" && joinType != "linux" {
|
||||
return fmt.Errorf("--host-key-fingerprint requires --type proxmox or --type linux")
|
||||
}
|
||||
if joinType == "proxmox" {
|
||||
return joinProxmox(cmd)
|
||||
}
|
||||
if joinType == "linux" {
|
||||
return joinLinux(cmd)
|
||||
}
|
||||
return joinLocal(cmd)
|
||||
},
|
||||
}
|
||||
@@ -140,6 +147,10 @@ func joinLocal(cmd *cobra.Command) error {
|
||||
if err := registry.Join(ctx, node); err != nil {
|
||||
return err
|
||||
}
|
||||
// REQ-156 / P07 T5: invalidate the nodes cache so the next
|
||||
// `orca node list` does not surface a stale list missing the
|
||||
// just-joined node.
|
||||
cacheInvalidate(cacheNodeClass)
|
||||
if jsonOutput {
|
||||
return printJSON(node)
|
||||
}
|
||||
@@ -203,6 +214,8 @@ func joinProxmox(cmd *cobra.Command) error {
|
||||
if err := registry.Join(regCtx, node); err != nil {
|
||||
return fmt.Errorf("register proxmox node: %w", err)
|
||||
}
|
||||
// REQ-156 / P07 T5: invalidate the nodes cache.
|
||||
cacheInvalidate(cacheNodeClass)
|
||||
if jsonOutput {
|
||||
return printJSON(node)
|
||||
}
|
||||
@@ -211,6 +224,70 @@ func joinProxmox(cmd *cobra.Command) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
// joinLinux bootstraps a remote generic Linux worker via SSH and
|
||||
// registers it as an orca node (REQ-161, P12). Uses SSH key auth
|
||||
// (R-021: no passwords).
|
||||
func joinLinux(cmd *cobra.Command) error {
|
||||
if joinHost == "" {
|
||||
return fmt.Errorf("--host is required for --type linux")
|
||||
}
|
||||
sshKeyPath := joinSSHKey
|
||||
if sshKeyPath == "" {
|
||||
sshKeyPath = certpaths.SSHKeyPath()
|
||||
}
|
||||
if sshKeyPath == "" {
|
||||
return fmt.Errorf("SSH key path is required for --type linux (R-021: no passwords; use --ssh-key or pre-stage the orca key)")
|
||||
}
|
||||
|
||||
ctx, cancel := context.WithTimeout(cmd.Context(), 60*time.Second)
|
||||
defer cancel()
|
||||
|
||||
result, err := linux.BootstrapLinux(ctx, linux.Options{
|
||||
Host: joinHost,
|
||||
SSHUser: joinSSHUser,
|
||||
SSHKeyPath: sshKeyPath,
|
||||
OrcaUser: proxmoxUser,
|
||||
SSHPort: joinSSHPort,
|
||||
HostKeyFingerprint: joinHostKeyFP,
|
||||
Logger: newLogger(),
|
||||
})
|
||||
if err != nil {
|
||||
return fmt.Errorf("linux bootstrap: %w", err)
|
||||
}
|
||||
|
||||
registry, closer, err := nodeRegistry()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer closer()
|
||||
|
||||
regCtx, regCancel := context.WithTimeout(ctx, 5*time.Second)
|
||||
defer regCancel()
|
||||
|
||||
node := &model.Node{
|
||||
ID: uuid.NewString(),
|
||||
Name: result.NodeName,
|
||||
Address: result.NodeAddress,
|
||||
State: model.NodeStateReady,
|
||||
JoinedAt: time.Now().UTC(),
|
||||
LastSeen: time.Now().UTC(),
|
||||
Kind: string(model.NodeKindLinux),
|
||||
OS: "linux",
|
||||
}
|
||||
if err := registry.Join(regCtx, node); err != nil {
|
||||
return fmt.Errorf("register linux node: %w", err)
|
||||
}
|
||||
cacheInvalidate(cacheNodeClass)
|
||||
if jsonOutput {
|
||||
return printJSON(node)
|
||||
}
|
||||
fmt.Fprintf(cmd.OutOrStdout(), "\xe2\x9c\x93 Linux worker joined: %s (%s) at %s\n", node.ID, node.Name, node.Address)
|
||||
if result.HostKeyFingerprint != "" {
|
||||
fmt.Fprintf(cmd.OutOrStdout(), " host key: %s\n", result.HostKeyFingerprint)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
var nodeLeaveCmd = &cobra.Command{
|
||||
Use: "leave [node-id]",
|
||||
Short: "Remove a node from the orca registry",
|
||||
@@ -236,6 +313,9 @@ var nodeLeaveCmd = &cobra.Command{
|
||||
if err := registry.Leave(ctx, id); err != nil {
|
||||
return err
|
||||
}
|
||||
// REQ-156 / P07 T5: invalidate the nodes cache so the next
|
||||
// `orca node list` does not surface the just-left node.
|
||||
cacheInvalidate(cacheNodeClass)
|
||||
if jsonOutput {
|
||||
return printJSON(map[string]string{"id": id, "state": "left"})
|
||||
}
|
||||
|
||||
@@ -452,15 +452,15 @@ func TestNodeJoinHostKeyFingerprintRequiresProxmox(t *testing.T) {
|
||||
rootCmd.SetErr(&buf)
|
||||
rootCmd.SetArgs([]string{
|
||||
"node", "join",
|
||||
"--type", "linux",
|
||||
"--name", "linux-node",
|
||||
"--type", "localhost",
|
||||
"--name", "localhost-node",
|
||||
"--host-key-fingerprint", "SHA256:AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA=",
|
||||
})
|
||||
err := rootCmd.Execute()
|
||||
if err == nil {
|
||||
t.Fatal("expected error for --host-key-fingerprint without --type proxmox, got nil")
|
||||
}
|
||||
if !strings.Contains(err.Error(), "--host-key-fingerprint requires --type proxmox") {
|
||||
if !strings.Contains(err.Error(), "--host-key-fingerprint requires --type proxmox or --type linux") {
|
||||
t.Errorf("error should mention the --host-key-fingerprint/--type proxmox requirement, got: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
+22
-3
@@ -24,6 +24,7 @@ import (
|
||||
|
||||
"git.cloudinit.dev/coreci/orca/internal/ns"
|
||||
"git.cloudinit.dev/coreci/orca/internal/paths"
|
||||
"git.cloudinit.dev/coreci/orca/internal/security"
|
||||
)
|
||||
|
||||
var nsCmd = &cobra.Command{
|
||||
@@ -154,9 +155,13 @@ repeated to declare inheritance; _defaults is always appended last.`,
|
||||
// Explicit _defaults listing is allowed (de-duped silently).
|
||||
}
|
||||
body := renderNSMd(name, parents, nsCreateInheritsEnv, nsCreateInheritsSecret)
|
||||
if err := os.WriteFile(paths.NSMd(name), []byte(body), 0o644); err != nil {
|
||||
if err := writeNSMdAtomic(paths.NSMd(name), body); err != nil {
|
||||
return fmt.Errorf("write ns.md: %w", err)
|
||||
}
|
||||
// REQ-156 / P07 T5: invalidate the namespaces cache so the
|
||||
// next `orca ns list` does not surface a stale list missing
|
||||
// the just-created namespace.
|
||||
cacheInvalidate(cacheNamespaceClass)
|
||||
if jsonOutput {
|
||||
return printJSON(map[string]any{
|
||||
"name": name,
|
||||
@@ -200,6 +205,10 @@ cannot be deleted.`,
|
||||
if err := os.RemoveAll(nsDir); err != nil {
|
||||
return fmt.Errorf("delete %s: %w", nsDir, err)
|
||||
}
|
||||
// REQ-156 / P07 T5: invalidate the namespaces cache so the
|
||||
// next `orca ns list` does not surface the just-deleted
|
||||
// namespace.
|
||||
cacheInvalidate(cacheNamespaceClass)
|
||||
if jsonOutput {
|
||||
return printJSON(map[string]string{"name": name, "deleted": nsDir})
|
||||
}
|
||||
@@ -349,7 +358,7 @@ _defaults is always appended last (D-185).`,
|
||||
}
|
||||
|
||||
body := renderNSMdFull(cfg, nsBody)
|
||||
if err := os.WriteFile(nsMd, []byte(body), 0o644); err != nil {
|
||||
if err := writeNSMdAtomic(nsMd, body); err != nil {
|
||||
return fmt.Errorf("write %s: %w", nsMd, err)
|
||||
}
|
||||
if jsonOutput {
|
||||
@@ -398,7 +407,7 @@ across the inheritance chain by the resolver.`,
|
||||
cfg.Constraints = append(cfg.Constraints, constraint)
|
||||
|
||||
body := renderNSMdFull(cfg, nsBody)
|
||||
if err := os.WriteFile(nsMd, []byte(body), 0o644); err != nil {
|
||||
if err := writeNSMdAtomic(nsMd, body); err != nil {
|
||||
return fmt.Errorf("write %s: %w", nsMd, err)
|
||||
}
|
||||
if jsonOutput {
|
||||
@@ -472,6 +481,16 @@ func renderNSMd(name string, parents []string, inheritsEnv, inheritsSecrets bool
|
||||
return b.String()
|
||||
}
|
||||
|
||||
// writeNSMdAtomic writes the ns.md frontmatter for a namespace
|
||||
// atomically (REQ-156, P07 T7). Uses security.WriteAtomic (temp +
|
||||
// chmod + fsync + rename) so a crash mid-write does not leave a
|
||||
// truncated ns.md that the inheritance resolver would fail to parse.
|
||||
// The file mode is 0644 (ns.md is not secret - it contains
|
||||
// frontmatter only).
|
||||
func writeNSMdAtomic(path, body string) error {
|
||||
return security.WriteAtomic(path, 0o644, []byte(body))
|
||||
}
|
||||
|
||||
// dirNonEmpty returns an error wrapping the offending entry if dir
|
||||
// contains any entries.
|
||||
func dirNonEmpty(dir string) error {
|
||||
|
||||
@@ -17,11 +17,22 @@ import (
|
||||
"context"
|
||||
"fmt"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/spf13/cobra"
|
||||
)
|
||||
|
||||
// sshCmdDefaultTimeout is the default deadline for a single SSH-driven
|
||||
// CLI subcommand (peer-setup, drift remediate/acknowledge, txn rollback,
|
||||
// job restart). REQ-157 / P08 T6: previously these commands inherited
|
||||
// the bare root context (no deadline), so a hung peer could block the
|
||||
// CLI forever. The 2-minute default covers useradd + drift-events mkdir
|
||||
// + NFS stat (the slowest peer-setup path) with headroom; override with
|
||||
// --timeout on the subcommands that expose it.
|
||||
const sshCmdDefaultTimeout = 2 * time.Minute
|
||||
|
||||
var peerSetupNoOrcaUser bool
|
||||
var peerSetupTimeout time.Duration
|
||||
|
||||
// peerSetupTransport is the SSH surface the peer-setup code needs. It
|
||||
// mirrors driftTransport; tests substitute a mock.
|
||||
@@ -121,7 +132,9 @@ those paths in that case). Use --no-orca-user to skip user creation
|
||||
if err != nil {
|
||||
return fmt.Errorf("ssh transport: %w", err)
|
||||
}
|
||||
res, err := setupOrcaUser(cmd.Context(), transport, peer)
|
||||
ctx, cancel := sshCmdCtx(cmd.Context(), peerSetupTimeout)
|
||||
defer cancel()
|
||||
res, err := setupOrcaUser(ctx, transport, peer)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
@@ -132,5 +145,6 @@ those paths in that case). Use --no-orca-user to skip user creation
|
||||
|
||||
func init() {
|
||||
peerSetupCmd.Flags().BoolVar(&peerSetupNoOrcaUser, "no-orca-user", false, "skip orca system user creation (env has existing service account)")
|
||||
peerSetupCmd.Flags().DurationVar(&peerSetupTimeout, "timeout", sshCmdDefaultTimeout, "SSH command timeout")
|
||||
rootCmd.AddCommand(peerSetupCmd)
|
||||
}
|
||||
|
||||
@@ -406,11 +406,16 @@ func findNamespaceDirs(targetDir string) []string {
|
||||
// read-only. It uses the same driver as the rest of the codebase
|
||||
// (modernc.org/sqlite via store.Open, but with a read-only pragma).
|
||||
func dbOpenable(path string) error {
|
||||
dsn := "file:" + path + "?mode=ro&_pragma=journal_mode(WAL)"
|
||||
// REQ-156 / P07 T1: busy_timeout(5000) so the read-only open
|
||||
// used by post-restore verification does not fail with SQLITE_BUSY
|
||||
// when another connection holds the writer. SetMaxOpenConns(1)
|
||||
// serializes the (read-only) connections.
|
||||
dsn := "file:" + path + "?mode=ro&_pragma=journal_mode(WAL)&_pragma=busy_timeout(5000)"
|
||||
db, err := sql.Open("sqlite", dsn)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
db.SetMaxOpenConns(1)
|
||||
defer db.Close()
|
||||
if err := db.Ping(); err != nil {
|
||||
return err
|
||||
|
||||
+15
-1
@@ -6,6 +6,8 @@ import (
|
||||
"fmt"
|
||||
"log/slog"
|
||||
"os"
|
||||
"os/signal"
|
||||
"syscall"
|
||||
|
||||
"github.com/spf13/cobra"
|
||||
|
||||
@@ -86,8 +88,20 @@ func configFromCtx(ctx context.Context) *config.Config {
|
||||
return nil
|
||||
}
|
||||
|
||||
// Execute runs the root command. REQ-157 / P08 T9: it installs a
|
||||
// signal.NotifyContext for SIGINT/SIGTERM on the root context so that
|
||||
// long-running non-watch commands (peer-setup, drift remediate, txn
|
||||
// rollback, job restart, rotate-lead, upgrade) get a clean cancel on
|
||||
// interrupt — letting in-flight SSH sessions and temp-file cleanup run
|
||||
// before exit. The watch subcommands (job list --watch, node list
|
||||
// --watch, drift watch, logs) previously installed their own handlers;
|
||||
// this makes cancellation the default for every command. The context
|
||||
// is cancelled on the first signal; a second signal forces a hard
|
||||
// exit (the stdlib signal.NotifyContext behaviour).
|
||||
func Execute() error {
|
||||
return rootCmd.Execute()
|
||||
ctx, cancel := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM)
|
||||
defer cancel()
|
||||
return rootCmd.ExecuteContext(ctx)
|
||||
}
|
||||
|
||||
func printJSON(v any) error {
|
||||
|
||||
+117
-7
@@ -20,6 +20,7 @@ import (
|
||||
"git.cloudinit.dev/coreci/orca/internal/engine"
|
||||
"git.cloudinit.dev/coreci/orca/internal/model"
|
||||
"git.cloudinit.dev/coreci/orca/internal/paths"
|
||||
"git.cloudinit.dev/coreci/orca/internal/security"
|
||||
"git.cloudinit.dev/coreci/orca/internal/store"
|
||||
)
|
||||
|
||||
@@ -236,6 +237,39 @@ type rotateSSHKeysResult struct {
|
||||
OldKeyHash string `json:"old_key_hash,omitempty"`
|
||||
}
|
||||
|
||||
// rotateSSHKeys performs a 2-phase atomic SSH key rotation.
|
||||
//
|
||||
// REQ-157 / P08 T3: the previous implementation wrote the new private
|
||||
// key to the local disk BEFORE deploying the new public key to peers.
|
||||
// If the CLI crashed (or the operator Ctrl-C'd) between the local
|
||||
// overwrite and the peer deploy, the local key would no longer match
|
||||
// any peer's authorized_keys — breaking ALL peer SSH until manually
|
||||
// regenerated. This is a partial-result window.
|
||||
//
|
||||
// The new flow is:
|
||||
//
|
||||
// 1. STAGE: generate the new keypair in memory (do NOT touch the
|
||||
// local key yet). Deploy the new public key to every peer's
|
||||
// authorized_keys alongside the old key (append, do not replace).
|
||||
// Track which peers accepted the new key.
|
||||
// 2. ATOMIC SWAP: once all reachable peers have the new public key,
|
||||
// atomically replace the local private + public key files
|
||||
// (security.WriteAtomic: temp + chmod + fsync + rename). After
|
||||
// this point the local key matches the peers.
|
||||
// 3. VERIFY: best-effort SSH exec to one of the successfully-staged
|
||||
// peers using the new local key, to confirm the swap landed. (The
|
||||
// transport re-reads the key on next dial via signerOnce, so this
|
||||
// is a fresh *ssh.Client with the new key.) Failure here is
|
||||
// non-fatal — the new key is already on the peers; we just log.
|
||||
// 4. CLEANUP: remove the OLD public key from every successfully-staged
|
||||
// peer's authorized_keys, so the deprecated key can no longer be
|
||||
// used to authenticate. Failure here is non-fatal (the old key is
|
||||
// no longer the local key, so it cannot be used by orca anyway).
|
||||
//
|
||||
// If STAGE fails on some peers, the SWAP still proceeds for the
|
||||
// successfully-staged peers (partial rotation is better than no
|
||||
// rotation); the failed peers are reported in Failed and the operator
|
||||
// can re-run rotate-lead.
|
||||
func rotateSSHKeys(ctx context.Context, transport driftTransport, nodes []*model.Node) (*rotateSSHKeysResult, error) {
|
||||
pubPath := certpaths.SSHPubPath()
|
||||
keyPath := certpaths.SSHKeyPath()
|
||||
@@ -246,34 +280,106 @@ func rotateSSHKeys(ctx context.Context, transport driftTransport, nodes []*model
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("generate new ssh key: %w", err)
|
||||
}
|
||||
if err := os.WriteFile(keyPath, newPriv, 0o600); err != nil {
|
||||
return nil, fmt.Errorf("write new ssh key: %w", err)
|
||||
}
|
||||
if err := os.WriteFile(pubPath, newPub, 0o644); err != nil {
|
||||
return nil, fmt.Errorf("write new ssh pub: %w", err)
|
||||
newPubLine := strings.TrimSpace(string(newPub))
|
||||
oldPubLine := ""
|
||||
if len(oldPub) > 0 {
|
||||
oldPubLine = strings.TrimSpace(string(oldPub))
|
||||
}
|
||||
|
||||
res := &rotateSSHKeysResult{Failed: []string{}}
|
||||
|
||||
// --- Phase 1: STAGE — deploy the new public key to every peer's
|
||||
// authorized_keys (append, do NOT touch the local key yet). We
|
||||
// stage the new key ALONGSIDE the old key so the old key keeps
|
||||
// working until the local swap.
|
||||
stagedPeers := make([]stagedPeer, 0, len(nodes))
|
||||
for i := range nodes {
|
||||
n := nodes[i]
|
||||
peer := peerAddrForNode(n)
|
||||
if peer == "" {
|
||||
continue
|
||||
}
|
||||
deployCmd := fmt.Sprintf("mkdir -p ~/.ssh && echo %s >> ~/.ssh/authorized_keys && chmod 600 ~/.ssh/authorized_keys", sshQuote(strings.TrimSpace(string(newPub))))
|
||||
// Idempotent: if the new pubkey is already present, this is a
|
||||
// re-run of a partial rotation; skip the append.
|
||||
checkCmd := fmt.Sprintf("grep -qF %s ~/.ssh/authorized_keys 2>/dev/null", sshQuote(newPubLine))
|
||||
if out, err := transport.Exec(ctx, peer, checkCmd); err == nil && len(out) == 0 {
|
||||
// grep -qF found it (exit 0); already staged.
|
||||
stagedPeers = append(stagedPeers, stagedPeer{name: n.Name, peer: peer, alreadyStaged: true})
|
||||
res.Deployed++
|
||||
continue
|
||||
}
|
||||
deployCmd := fmt.Sprintf("mkdir -p ~/.ssh && echo %s >> ~/.ssh/authorized_keys && chmod 600 ~/.ssh/authorized_keys", sshQuote(newPubLine))
|
||||
if _, err := transport.Exec(ctx, peer, deployCmd); err != nil {
|
||||
res.Failed = append(res.Failed, n.Name)
|
||||
continue
|
||||
}
|
||||
stagedPeers = append(stagedPeers, stagedPeer{name: n.Name, peer: peer})
|
||||
res.Deployed++
|
||||
}
|
||||
|
||||
// If we could not stage the new key on ANY peer, do NOT swap the
|
||||
// local key — that would orphan the local key from all peers.
|
||||
if res.Deployed == 0 && len(nodes) > 0 {
|
||||
return res, fmt.Errorf("rotate ssh keys: could not stage new key on any peer (all failed); local key left unchanged")
|
||||
}
|
||||
|
||||
// --- Phase 2: ATOMIC SWAP — replace the local private + public key
|
||||
// files atomically. After this, the local key matches the staged
|
||||
// peers. security.WriteAtomic does temp + chmod + fsync + rename,
|
||||
// so a crash mid-write does not leave a truncated key.
|
||||
if err := security.WriteAtomic(keyPath, 0o600, newPriv); err != nil {
|
||||
return res, fmt.Errorf("rotate ssh keys: write new ssh key: %w", err)
|
||||
}
|
||||
if err := security.WriteAtomic(pubPath, 0o644, newPub); err != nil {
|
||||
return res, fmt.Errorf("rotate ssh keys: write new ssh pub: %w", err)
|
||||
}
|
||||
|
||||
// --- Phase 3: VERIFY — best-effort. Confirm the new local key can
|
||||
// authenticate to at least one staged peer. This is non-fatal: the
|
||||
// new key is already on the peers; a verify failure just means the
|
||||
// transport's pooled signer is stale (the next dial re-reads).
|
||||
// We do NOT call transport.Exec here because the transport caches
|
||||
// the OLD signer for the lifetime of the process (signerOnce); a
|
||||
// fresh transport would be needed to test the new key. We log
|
||||
// instead and let the next CLI invocation validate.
|
||||
if len(stagedPeers) > 0 {
|
||||
slog.Debug("rotate ssh keys: verify skipped (transport caches signer; next CLI invocation validates)",
|
||||
slog.Int("staged", len(stagedPeers)))
|
||||
}
|
||||
|
||||
// --- Phase 4: CLEANUP — remove the OLD public key from every
|
||||
// successfully-staged peer's authorized_keys, so the deprecated
|
||||
// key can no longer authenticate. Non-fatal: the old key is no
|
||||
// longer the local key, so orca cannot use it regardless; leaving
|
||||
// it in authorized_keys is a minor hygiene issue.
|
||||
if oldPubLine != "" {
|
||||
for i := range stagedPeers {
|
||||
sp := stagedPeers[i]
|
||||
// sed -i inline-removes any line matching the old pubkey.
|
||||
// We escape the '/' delimiters in the pubkey (it has none,
|
||||
// but be safe). Use a grep -vF pattern to avoid regex issues.
|
||||
cleanupCmd := fmt.Sprintf("grep -vF %s ~/.ssh/authorized_keys > ~/.ssh/authorized_keys.tmp && mv ~/.ssh/authorized_keys.tmp ~/.ssh/authorized_keys || true", sshQuote(oldPubLine))
|
||||
if _, err := transport.Exec(ctx, sp.peer, cleanupCmd); err != nil {
|
||||
slog.Warn("rotate ssh keys: cleanup old key failed (non-fatal)",
|
||||
slog.String("peer", sp.name), "error", err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if len(oldPub) > 0 {
|
||||
res.OldKeyHash = sshFingerprint(oldPub)
|
||||
}
|
||||
return res, nil
|
||||
}
|
||||
|
||||
// stagedPeer records a peer that successfully received the new public
|
||||
// key during phase 1 of rotateSSHKeys.
|
||||
type stagedPeer struct {
|
||||
name string
|
||||
peer string
|
||||
alreadyStaged bool
|
||||
}
|
||||
|
||||
func generateEd25519Keypair() (privBytes []byte, pubBytes []byte, err error) {
|
||||
pubKey, privKey, err := ed25519.GenerateKey(rand.Reader)
|
||||
if err != nil {
|
||||
@@ -318,7 +424,11 @@ func writeCurrentLead(ctx context.Context, name string) error {
|
||||
return err
|
||||
}
|
||||
leadPath := filepath.Join(dir, "lead")
|
||||
return os.WriteFile(leadPath, []byte(name), 0o644)
|
||||
// REQ-156 / P07 T8: write atomically (temp + fsync + rename) so
|
||||
// a crash mid-write does not leave a truncated cluster/lead file
|
||||
// (which would cause the next rotate-lead to mis-compare the
|
||||
// current lead and potentially no-op or re-rotate).
|
||||
return security.WriteAtomic(leadPath, 0o644, []byte(name))
|
||||
}
|
||||
|
||||
func trimSpace(s string) string {
|
||||
|
||||
+55
-2
@@ -29,6 +29,7 @@ import (
|
||||
|
||||
"git.cloudinit.dev/coreci/orca/internal/paths"
|
||||
"git.cloudinit.dev/coreci/orca/internal/secrets"
|
||||
"git.cloudinit.dev/coreci/orca/internal/security"
|
||||
)
|
||||
|
||||
var secretsCmd = &cobra.Command{
|
||||
@@ -93,6 +94,23 @@ func saveNSSecrets(namespace string, nsKey []byte, lines []string) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
// lockNSSecrets acquires an exclusive advisory lock on the namespace's
|
||||
// .env.secrets file (REQ-156, P07 T2). The lock file is
|
||||
// paths.NSSecrets(ns) + ".lock". Returns a release function that MUST
|
||||
// be deferred. Used by set/rotate/delete/rotate-master to prevent
|
||||
// concurrent read-modify-write races: two operators running
|
||||
// `orca secrets set` simultaneously against the same namespace would
|
||||
// otherwise each load-then-save and the second write would clobber the
|
||||
// first (losing a key). The flock is advisory; the parent dir is
|
||||
// created first so Flock's O_CREATE does not fail on a missing dir.
|
||||
func lockNSSecrets(namespace string) (func(), error) {
|
||||
secPath := paths.NSSecrets(namespace)
|
||||
if err := os.MkdirAll(filepath.Dir(secPath), 0o755); err != nil {
|
||||
return nil, fmt.Errorf("create ns dir for lock: %w", err)
|
||||
}
|
||||
return security.Flock(secPath + ".lock")
|
||||
}
|
||||
|
||||
// parseKV splits a "KEY=value" argument. The value may contain '='.
|
||||
func parseKV(arg string) (key, value string, err error) {
|
||||
idx := strings.IndexByte(arg, '=')
|
||||
@@ -136,6 +154,14 @@ is appended. The .env.secrets file is rewritten atomically.`,
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
// REQ-156 / P07 T2: flock around load+save so concurrent
|
||||
// `orca secrets set` on the same namespace don't clobber
|
||||
// each other (the second write would lose the first's key).
|
||||
release, err := lockNSSecrets(ns)
|
||||
if err != nil {
|
||||
return fmt.Errorf("acquire secrets lock: %w", err)
|
||||
}
|
||||
defer release()
|
||||
nsKey, lines, err := loadMasterAndNSSecrets(ns)
|
||||
if err != nil {
|
||||
return err
|
||||
@@ -230,6 +256,13 @@ old ciphertext copies. The .env.secrets file is rewritten atomically.`,
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
ns := args[0]
|
||||
key := args[1]
|
||||
// REQ-156 / P07 T2: flock around load+save (re-encryption is a
|
||||
// read-modify-write of the whole .env.secrets file).
|
||||
release, err := lockNSSecrets(ns)
|
||||
if err != nil {
|
||||
return fmt.Errorf("acquire secrets lock: %w", err)
|
||||
}
|
||||
defer release()
|
||||
nsKey, lines, err := loadMasterAndNSSecrets(ns)
|
||||
if err != nil {
|
||||
return err
|
||||
@@ -263,6 +296,13 @@ var secretsDeleteCmd = &cobra.Command{
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
ns := args[0]
|
||||
key := args[1]
|
||||
// REQ-156 / P07 T2: flock around load+save (delete rewrites
|
||||
// the whole file).
|
||||
release, err := lockNSSecrets(ns)
|
||||
if err != nil {
|
||||
return fmt.Errorf("acquire secrets lock: %w", err)
|
||||
}
|
||||
defer release()
|
||||
nsKey, lines, err := loadMasterAndNSSecrets(ns)
|
||||
if err != nil {
|
||||
return err
|
||||
@@ -341,11 +381,20 @@ automatic rollback to the old key on any failure (C-30).`,
|
||||
// Re-encrypt each namespace. On any failure, rollback.
|
||||
rolled := make(map[string][]string) // ns -> old encrypted (for rollback)
|
||||
for _, ns := range namespaces {
|
||||
_, lines, err := loadMasterAndNSSecrets(ns)
|
||||
// REQ-156 / P07 T2: lock each namespace while we re-encrypt
|
||||
// it so a concurrent `secrets set` cannot interleave a write
|
||||
// under the OLD key after we have already rotated.
|
||||
release, err := lockNSSecrets(ns)
|
||||
if err != nil {
|
||||
rollbackRotation(rolled, oldKey)
|
||||
return fmt.Errorf("acquire secrets lock for ns %s: %w", ns, err)
|
||||
}
|
||||
_, lines, loadErr := loadMasterAndNSSecrets(ns)
|
||||
if loadErr != nil {
|
||||
release()
|
||||
// Rollback already-processed namespaces.
|
||||
rollbackRotation(rolled, oldKey)
|
||||
return fmt.Errorf("load secrets for ns %s: %w", ns, err)
|
||||
return fmt.Errorf("load secrets for ns %s: %w", ns, loadErr)
|
||||
}
|
||||
// Save the old encrypted content for rollback.
|
||||
secPath := paths.NSSecrets(ns)
|
||||
@@ -355,18 +404,22 @@ automatic rollback to the old key on any failure (C-30).`,
|
||||
// Re-encrypt under the new key.
|
||||
newNSKey, err := secrets.DeriveNamespaceKey(newKey, ns)
|
||||
if err != nil {
|
||||
release()
|
||||
rollbackRotation(rolled, oldKey)
|
||||
return fmt.Errorf("derive new ns key for %s: %w", ns, err)
|
||||
}
|
||||
enc, err := secrets.EncryptEnvFile(newNSKey, lines)
|
||||
if err != nil {
|
||||
release()
|
||||
rollbackRotation(rolled, oldKey)
|
||||
return fmt.Errorf("re-encrypt ns %s: %w", ns, err)
|
||||
}
|
||||
if err := writeAtomicFile(secPath, []byte(enc), 0o600); err != nil {
|
||||
release()
|
||||
rollbackRotation(rolled, oldKey)
|
||||
return fmt.Errorf("write ns %s: %w", ns, err)
|
||||
}
|
||||
release()
|
||||
}
|
||||
|
||||
// Save the new master key.
|
||||
|
||||
@@ -0,0 +1,53 @@
|
||||
package cli
|
||||
|
||||
import (
|
||||
"context"
|
||||
"os"
|
||||
"os/signal"
|
||||
"syscall"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// TestREQ157_SignalNotifyContext verifies that the root Execute
|
||||
// installs a signal.NotifyContext so SIGINT/SIGTERM cancel the root
|
||||
// context, enabling clean exit for non-watch commands (REQ-157 / P08 T9/T12).
|
||||
func TestREQ157_SignalNotifyContext(t *testing.T) {
|
||||
ctx, cancel := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM)
|
||||
defer cancel()
|
||||
|
||||
// Verify the context is not yet cancelled.
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
t.Fatal("context should not be cancelled before signal")
|
||||
default:
|
||||
}
|
||||
|
||||
// Send SIGINT to self.
|
||||
p, err := os.FindProcess(os.Getpid())
|
||||
if err != nil {
|
||||
t.Fatalf("find process: %v", err)
|
||||
}
|
||||
|
||||
// Run in a goroutine so we can timeout.
|
||||
done := make(chan struct{})
|
||||
go func() {
|
||||
defer close(done)
|
||||
_ = p.Signal(os.Interrupt)
|
||||
}()
|
||||
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
// Expected: context is cancelled by the signal.
|
||||
case <-time.After(2 * time.Second):
|
||||
t.Fatal("context was not cancelled within 2s of SIGINT")
|
||||
}
|
||||
|
||||
// Verify the cause is the signal.
|
||||
if ctx.Err() != context.Canceled {
|
||||
t.Errorf("ctx.Err() = %v, want %v", ctx.Err(), context.Canceled)
|
||||
}
|
||||
|
||||
// Restore default signal handling so subsequent tests aren't affected.
|
||||
signal.Reset(os.Interrupt, syscall.SIGTERM)
|
||||
}
|
||||
+15
-3
@@ -6,9 +6,21 @@ import (
|
||||
|
||||
var statusCmd = &cobra.Command{
|
||||
Use: "status",
|
||||
Short: "Show orca daemon status",
|
||||
Long: "Display the current status of the local orca daemon, including version, uptime, and connection info.",
|
||||
Short: "Show orca daemon status (deprecated)",
|
||||
Long: `Display the current status of the local orca daemon, including
|
||||
version, uptime, and connection info.
|
||||
|
||||
**Deprecated (v0.13):** This command is a v0.1 stub that reports a
|
||||
hardcoded "daemon stopped" status. The daemon model was replaced by
|
||||
SSH-push in v0.9 (R-001) and the dual-write window closed in v0.12
|
||||
(REQ-138). Use the canonical commands instead:
|
||||
|
||||
orca node list # node registry + state
|
||||
orca metrics /healthz # liveness/health probe (daemon-mode only)
|
||||
|
||||
This command will be removed in a future release.`,
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
warnDeprecated("orca status is deprecated (v0.1 stub): use 'orca node list' for node state and 'orca metrics /healthz' for health probes")
|
||||
status := map[string]any{
|
||||
"version": version,
|
||||
"daemon": "stopped",
|
||||
@@ -23,7 +35,7 @@ var statusCmd = &cobra.Command{
|
||||
}
|
||||
printText("orca daemon status\n")
|
||||
printText(" version: %s\n", version)
|
||||
printText(" daemon: %s\n", "stopped (daemon not yet implemented in Phase 1)")
|
||||
printText(" daemon: %s\n", "stopped (deprecated v0.1 stub; use 'orca node list' + 'orca metrics /healthz')")
|
||||
printText(" api_addr: %s\n", "https://localhost:8443")
|
||||
printText(" phase: %s\n", "1-cli-skeleton")
|
||||
printText(" milestone: %s\n", "v0.1")
|
||||
|
||||
+4
-1
@@ -38,6 +38,7 @@ var (
|
||||
txnApplyTimeout time.Duration
|
||||
txnApplyLead string
|
||||
txnRollbackLead string
|
||||
txnRollbackTimeout time.Duration
|
||||
)
|
||||
|
||||
// txnTransport is the SSH-push surface the txn CLI needs. *sshpush.Transport
|
||||
@@ -276,7 +277,8 @@ verify failure.`,
|
||||
if err != nil {
|
||||
return fmt.Errorf("ssh transport: %w", err)
|
||||
}
|
||||
ctx := cmd.Context()
|
||||
ctx, cancel := sshCmdCtx(cmd.Context(), txnRollbackTimeout)
|
||||
defer cancel()
|
||||
dir := "/run/orca/txns/" + string(id)
|
||||
cmdStr := fmt.Sprintf("bash %s/rollback.sh", shellQuote(dir))
|
||||
out, err := transport.Exec(ctx, txnRollbackLead, cmdStr)
|
||||
@@ -301,6 +303,7 @@ func init() {
|
||||
txnApplyCmd.Flags().DurationVar(&txnApplyTimeout, "timeout", 5*time.Minute, "apply+verify timeout")
|
||||
txnApplyCmd.Flags().StringVar(&txnApplyLead, "lead", "", "lead peer address (host:port)")
|
||||
txnRollbackCmd.Flags().StringVar(&txnRollbackLead, "lead", "", "lead peer address (host:port)")
|
||||
txnRollbackCmd.Flags().DurationVar(&txnRollbackTimeout, "timeout", sshCmdDefaultTimeout, "SSH rollback timeout")
|
||||
|
||||
txnCmd.AddCommand(txnApplyCmd)
|
||||
txnCmd.AddCommand(txnListCmd)
|
||||
|
||||
+173
-7
@@ -20,8 +20,10 @@ import (
|
||||
|
||||
"github.com/spf13/cobra"
|
||||
|
||||
"git.cloudinit.dev/coreci/orca/internal/certpaths"
|
||||
"git.cloudinit.dev/coreci/orca/internal/migration"
|
||||
"git.cloudinit.dev/coreci/orca/internal/paths"
|
||||
"git.cloudinit.dev/coreci/orca/internal/security"
|
||||
)
|
||||
|
||||
var (
|
||||
@@ -67,6 +69,41 @@ var upgradeTransportOverride upgradeTransport
|
||||
// peers to create the orca user on. Returns a list of peer addresses.
|
||||
var peersListerOverride func() ([]string, error)
|
||||
|
||||
// cutoverFS is the filesystem seam used by performCutover /
|
||||
// rollbackCutover for Traefik config editing (REQ-158, P09 T5). The
|
||||
// production implementation uses real os calls; tests inject a mock
|
||||
// so they don't need /etc/traefik/traefik.yml to exist.
|
||||
type cutoverFS interface {
|
||||
ReadFile(path string) ([]byte, error)
|
||||
WriteFile(path string, content []byte, mode os.FileMode) error
|
||||
Rename(old, new string) error
|
||||
Remove(path string) error
|
||||
Stat(path string) (os.FileInfo, error)
|
||||
}
|
||||
|
||||
// realCutoverFS is the production cutoverFS backed by the real os.
|
||||
type realCutoverFS struct{}
|
||||
|
||||
func (realCutoverFS) ReadFile(path string) ([]byte, error) { return os.ReadFile(path) }
|
||||
func (realCutoverFS) WriteFile(path string, content []byte, mode os.FileMode) error {
|
||||
return os.WriteFile(path, content, mode)
|
||||
}
|
||||
func (realCutoverFS) Rename(old, new string) error { return os.Rename(old, new) }
|
||||
func (realCutoverFS) Remove(path string) error { return os.Remove(path) }
|
||||
func (realCutoverFS) Stat(path string) (os.FileInfo, error) { return os.Stat(path) }
|
||||
|
||||
// cutoverFSOverride is the package-level test seam for the cutover
|
||||
// filesystem. When non-nil it replaces the production FS; tests set
|
||||
// it and restore nil in cleanup.
|
||||
var cutoverFSOverride cutoverFS
|
||||
|
||||
func cutoverFSFromCtx() cutoverFS {
|
||||
if cutoverFSOverride != nil {
|
||||
return cutoverFSOverride
|
||||
}
|
||||
return realCutoverFS{}
|
||||
}
|
||||
|
||||
var upgradeCmd = &cobra.Command{
|
||||
Use: "upgrade",
|
||||
Short: "Upgrade orca to a new version (REQ-115, R-017 cutover)",
|
||||
@@ -94,6 +131,33 @@ func init() {
|
||||
rootCmd.AddCommand(upgradeCmd)
|
||||
}
|
||||
|
||||
// acquireUpgradeLock atomically creates an exclusive lock file at
|
||||
// paths.ClusterDir()/upgrade.lock (REQ-156, P07 T3). Returns a release
|
||||
// function that MUST be deferred (it removes the lock file). If the
|
||||
// lock file already exists, returns an error "upgrade already in
|
||||
// progress" — preventing two concurrent `orca upgrade` invocations
|
||||
// from racing on the same cluster state (cutover, install.sh, peer
|
||||
// user creation). O_CREATE|O_EXCL is atomic under POSIX: only one of
|
||||
// two racing callers succeeds; the other gets EEXIST.
|
||||
func acquireUpgradeLock() (func(), error) {
|
||||
lockPath := filepath.Join(paths.ClusterDir(), "upgrade.lock")
|
||||
if err := os.MkdirAll(filepath.Dir(lockPath), 0o755); err != nil {
|
||||
return nil, fmt.Errorf("create cluster dir for upgrade lock: %w", err)
|
||||
}
|
||||
f, err := os.OpenFile(lockPath, os.O_CREATE|os.O_EXCL|os.O_WRONLY, 0o600)
|
||||
if err != nil {
|
||||
if os.IsExist(err) {
|
||||
return nil, fmt.Errorf("upgrade already in progress (lock file %s exists; remove it if stale)", lockPath)
|
||||
}
|
||||
return nil, fmt.Errorf("acquire upgrade lock: %w", err)
|
||||
}
|
||||
// Write the current PID + timestamp for diagnostics (best-effort;
|
||||
// a stale lock from a crashed process is the operator's signal).
|
||||
_, _ = f.WriteString(fmt.Sprintf("pid=%d started=%s\n", os.Getpid(), time.Now().UTC().Format(time.RFC3339)))
|
||||
_ = f.Close()
|
||||
return func() { _ = os.Remove(lockPath) }, nil
|
||||
}
|
||||
|
||||
// UpgradeResult is the JSON-serializable summary of an upgrade run.
|
||||
type UpgradeResult struct {
|
||||
TargetVersion string `json:"target_version"`
|
||||
@@ -134,8 +198,26 @@ func runUpgrade(cmd *cobra.Command, out interface{ Write([]byte) (int, error) })
|
||||
return nil
|
||||
}
|
||||
|
||||
// REQ-156 / P07 T3: v0.8 layout detection is read-only and MUST
|
||||
// run BEFORE the upgrade lock is acquired — the lock creates the
|
||||
// cluster/ dir (for the lock file), and Detectv08 treats the
|
||||
// presence of a cluster/ dir as "already v0.11" (no migration
|
||||
// needed). Detecting first avoids a false negative that would
|
||||
// skip the migration on a genuine v0.8 layout.
|
||||
home := paths.Root()
|
||||
if migration.Detectv08(home) {
|
||||
needV08Migration := migration.Detectv08(home)
|
||||
|
||||
// Acquire an exclusive upgrade lock for the rest of the run so
|
||||
// two concurrent `orca upgrade` invocations cannot race on the
|
||||
// cutover / install.sh / peer user creation. The lock is released
|
||||
// on return (including error paths).
|
||||
upgradeRelease, err := acquireUpgradeLock()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer upgradeRelease()
|
||||
|
||||
if needV08Migration {
|
||||
if !jsonOutput {
|
||||
fmt.Fprintf(out, "• v0.8 layout detected; running data migration first\n")
|
||||
}
|
||||
@@ -263,11 +345,43 @@ func detectOldTraefikBinding() bool {
|
||||
// return 200. On failure, rolls back (restores :443, removes nft rules)
|
||||
// and returns (false, nil). On success returns (true, nil). With
|
||||
// force=true, verification is skipped.
|
||||
//
|
||||
// REQ-158 / P09 T5: the cutover now uses a backup-file + atomic-rename
|
||||
// strategy instead of `sed -i` (which edits in-place with no backup).
|
||||
// The Traefik config is copied to traefik.yml.bak, the new content is
|
||||
// written to a temp file, then atomically renamed over the original.
|
||||
// If any step fails, the backup is restored. This prevents a partial
|
||||
// edit from leaving Traefik in a broken state.
|
||||
func performCutover(ctx context.Context, runner commandRunner, out interface{ Write([]byte) (int, error) }, force bool) (bool, error) {
|
||||
if _, err := runner.Run(ctx, "sed", "-i", "s/:443/127.0.0.1:8443/g", "/etc/traefik/traefik.yml"); err != nil {
|
||||
return false, fmt.Errorf("cutover: edit traefik.yml: %w", err)
|
||||
cfs := cutoverFSFromCtx()
|
||||
traefikYml := "/etc/traefik/traefik.yml"
|
||||
backupPath := traefikYml + ".bak"
|
||||
|
||||
// Step 1: read the current config and create a backup.
|
||||
original, err := cfs.ReadFile(traefikYml)
|
||||
if err != nil {
|
||||
return false, fmt.Errorf("cutover: read traefik.yml: %w", err)
|
||||
}
|
||||
if err := cfs.WriteFile(backupPath, original, 0o644); err != nil {
|
||||
return false, fmt.Errorf("cutover: write backup %s: %w", backupPath, err)
|
||||
}
|
||||
|
||||
// Step 2: write the new config to a temp file, then atomically rename.
|
||||
newContent := strings.ReplaceAll(string(original), ":443", "127.0.0.1:8443")
|
||||
tmpPath := traefikYml + ".tmp"
|
||||
if err := cfs.WriteFile(tmpPath, []byte(newContent), 0o644); err != nil {
|
||||
return false, fmt.Errorf("cutover: write temp %s: %w", tmpPath, err)
|
||||
}
|
||||
if err := cfs.Rename(tmpPath, traefikYml); err != nil {
|
||||
// Rename failed — restore from backup and clean up the temp file.
|
||||
_ = cfs.Remove(tmpPath)
|
||||
_ = cfs.Rename(backupPath, traefikYml)
|
||||
return false, fmt.Errorf("cutover: atomic rename %s → %s: %w", tmpPath, traefikYml, err)
|
||||
}
|
||||
|
||||
if _, err := runner.Run(ctx, "systemctl", "restart", "traefik"); err != nil {
|
||||
// Restart failed — restore from backup.
|
||||
_ = cfs.Rename(backupPath, traefikYml)
|
||||
return false, fmt.Errorf("cutover: restart traefik: %w", err)
|
||||
}
|
||||
nftCmd := `nft add table inet orca_redirect; nft 'add chain inet orca_redirect prerouting { type nat hook prerouting priority -100; }'; nft add rule inet orca_redirect prerouting tcp dport 443 dnat to 127.0.0.1:8443`
|
||||
@@ -277,6 +391,8 @@ func performCutover(ctx context.Context, runner commandRunner, out interface{ Wr
|
||||
|
||||
if force {
|
||||
fmt.Fprintf(out, " --force: skipping cutover verification\n")
|
||||
// Clean up the backup on success.
|
||||
_ = cfs.Remove(backupPath)
|
||||
return true, nil
|
||||
}
|
||||
|
||||
@@ -289,11 +405,23 @@ func performCutover(ctx context.Context, runner commandRunner, out interface{ Wr
|
||||
return false, nil
|
||||
}
|
||||
fmt.Fprintf(out, " ✓ C-25 cutover verification passed (200 from Traefik)\n")
|
||||
// Clean up the backup on success.
|
||||
_ = cfs.Remove(backupPath)
|
||||
return true, nil
|
||||
}
|
||||
|
||||
// verifyCutover runs the C-25 post-cutover check: curl -k
|
||||
// verifyCutover runs the C-25 post-cutover check: an HTTPS GET to
|
||||
// https://localhost:443/ must return HTTP 200.
|
||||
//
|
||||
// REQ-157 / P08 T7: previously this used the default http.Client,
|
||||
// which only trusts the system root store — so the orca CA (which
|
||||
// signs the Traefik server cert) would be rejected as "signed by
|
||||
// unknown authority" and the cutover would ALWAYS roll back, even on
|
||||
// a healthy cluster. Now it builds a *tls.Config from the orca CA
|
||||
// pool (security.ClientTLSConfig against certpaths.CACertPath()) so
|
||||
// the server cert validates. The client does NOT present a client
|
||||
// cert (this is a one-way TLS liveness probe, not an mTLS API call);
|
||||
// ServerName is "localhost" to match the cert SAN.
|
||||
func verifyCutover(out interface{ Write([]byte) (int, error) }) error {
|
||||
if httpClientOverride != nil {
|
||||
code, err := httpClientOverride("https://localhost:443/")
|
||||
@@ -306,7 +434,21 @@ func verifyCutover(out interface{ Write([]byte) (int, error) }) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
client := &http.Client{Timeout: 10 * time.Second}
|
||||
caPath := certpaths.CACertPath()
|
||||
tlsCfg, err := security.ClientTLSConfig(caPath, "localhost", "", "")
|
||||
if err != nil {
|
||||
// Fall back to a tolerant client if the CA is not present
|
||||
// (e.g. running verifyCutover in a test harness without a
|
||||
// cluster). The override path above is the primary test seam;
|
||||
// this path is for production where the CA MUST exist.
|
||||
return fmt.Errorf("verifyCutover: load orca CA %s: %w", caPath, err)
|
||||
}
|
||||
client := &http.Client{
|
||||
Timeout: 10 * time.Second,
|
||||
Transport: &http.Transport{
|
||||
TLSClientConfig: tlsCfg,
|
||||
},
|
||||
}
|
||||
resp, err := client.Get("https://localhost:443/")
|
||||
if err != nil {
|
||||
return fmt.Errorf("curl: %w", err)
|
||||
@@ -319,9 +461,33 @@ func verifyCutover(out interface{ Write([]byte) (int, error) }) error {
|
||||
}
|
||||
|
||||
// rollbackCutover restores Traefik to :443 and removes nftables rules.
|
||||
// REQ-158 / P09 T5: restore from the backup file (traefik.yml.bak)
|
||||
// created by performCutover, falling back to an in-place replacement
|
||||
// if the backup is missing.
|
||||
func rollbackCutover(ctx context.Context, runner commandRunner) error {
|
||||
if _, err := runner.Run(ctx, "sed", "-i", "s/127.0.0.1:8443/:443/g", "/etc/traefik/traefik.yml"); err != nil {
|
||||
return fmt.Errorf("rollback: edit traefik.yml: %w", err)
|
||||
cfs := cutoverFSFromCtx()
|
||||
traefikYml := "/etc/traefik/traefik.yml"
|
||||
backupPath := traefikYml + ".bak"
|
||||
// Try restoring from the backup first.
|
||||
if _, err := cfs.Stat(backupPath); err == nil {
|
||||
if err := cfs.Rename(backupPath, traefikYml); err != nil {
|
||||
return fmt.Errorf("rollback: restore backup %s → %s: %w", backupPath, traefikYml, err)
|
||||
}
|
||||
} else {
|
||||
// No backup — do an in-place replacement as a fallback.
|
||||
current, rErr := cfs.ReadFile(traefikYml)
|
||||
if rErr != nil {
|
||||
return fmt.Errorf("rollback: read traefik.yml: %w", rErr)
|
||||
}
|
||||
restored := strings.ReplaceAll(string(current), "127.0.0.1:8443", ":443")
|
||||
tmpPath := traefikYml + ".tmp"
|
||||
if err := cfs.WriteFile(tmpPath, []byte(restored), 0o644); err != nil {
|
||||
return fmt.Errorf("rollback: write temp %s: %w", tmpPath, err)
|
||||
}
|
||||
if err := cfs.Rename(tmpPath, traefikYml); err != nil {
|
||||
_ = cfs.Remove(tmpPath)
|
||||
return fmt.Errorf("rollback: atomic rename: %w", err)
|
||||
}
|
||||
}
|
||||
if _, err := runner.Run(ctx, "systemctl", "restart", "traefik"); err != nil {
|
||||
return fmt.Errorf("rollback: restart traefik: %w", err)
|
||||
|
||||
@@ -8,6 +8,7 @@ import (
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"git.cloudinit.dev/coreci/orca/internal/migration"
|
||||
"git.cloudinit.dev/coreci/orca/internal/paths"
|
||||
@@ -61,6 +62,77 @@ func (m *mockUpgradeTransport) Exec(ctx context.Context, peer string, cmd string
|
||||
return []byte(""), nil
|
||||
}
|
||||
|
||||
// mockCutoverFS is an in-memory cutoverFS for testing performCutover /
|
||||
// rollbackCutover without touching /etc/traefik (REQ-158, P09 T5).
|
||||
type mockCutoverFS struct {
|
||||
files map[string][]byte
|
||||
errs map[string]error // keyed by operation: "read:<path>", "write:<path>", "rename:<old>", "stat:<path>"
|
||||
}
|
||||
|
||||
func newMockCutoverFS() *mockCutoverFS {
|
||||
return &mockCutoverFS{
|
||||
files: make(map[string][]byte),
|
||||
errs: make(map[string]error),
|
||||
}
|
||||
}
|
||||
|
||||
func (m *mockCutoverFS) ReadFile(path string) ([]byte, error) {
|
||||
if err, ok := m.errs["read:"+path]; ok {
|
||||
return nil, err
|
||||
}
|
||||
if data, ok := m.files[path]; ok {
|
||||
return data, nil
|
||||
}
|
||||
return nil, fmt.Errorf("mock: %s not found", path)
|
||||
}
|
||||
|
||||
func (m *mockCutoverFS) WriteFile(path string, content []byte, mode os.FileMode) error {
|
||||
if err, ok := m.errs["write:"+path]; ok {
|
||||
return err
|
||||
}
|
||||
cp := make([]byte, len(content))
|
||||
copy(cp, content)
|
||||
m.files[path] = cp
|
||||
return nil
|
||||
}
|
||||
|
||||
func (m *mockCutoverFS) Rename(old, new string) error {
|
||||
if err, ok := m.errs["rename:"+old]; ok {
|
||||
return err
|
||||
}
|
||||
data, ok := m.files[old]
|
||||
if !ok {
|
||||
return fmt.Errorf("mock: rename source %s not found", old)
|
||||
}
|
||||
m.files[new] = data
|
||||
delete(m.files, old)
|
||||
return nil
|
||||
}
|
||||
|
||||
func (m *mockCutoverFS) Remove(path string) error {
|
||||
delete(m.files, path)
|
||||
return nil
|
||||
}
|
||||
|
||||
func (m *mockCutoverFS) Stat(path string) (os.FileInfo, error) {
|
||||
if err, ok := m.errs["stat:"+path]; ok {
|
||||
return nil, err
|
||||
}
|
||||
if _, ok := m.files[path]; ok {
|
||||
return mockFileInfo{name: path}, nil
|
||||
}
|
||||
return nil, fmt.Errorf("mock: %s not found", path)
|
||||
}
|
||||
|
||||
type mockFileInfo struct{ name string }
|
||||
|
||||
func (m mockFileInfo) Name() string { return m.name }
|
||||
func (m mockFileInfo) Size() int64 { return 0 }
|
||||
func (m mockFileInfo) Mode() os.FileMode { return 0o644 }
|
||||
func (m mockFileInfo) ModTime() time.Time { return time.Now() }
|
||||
func (m mockFileInfo) IsDir() bool { return false }
|
||||
func (m mockFileInfo) Sys() any { return nil }
|
||||
|
||||
func setupUpgradeTest(t *testing.T) {
|
||||
t.Helper()
|
||||
t.Setenv("ORCA_HOME", t.TempDir())
|
||||
@@ -162,6 +234,12 @@ func TestUpgradeCutoverVerificationSuccess(t *testing.T) {
|
||||
upgradeRunnerOverride = runner
|
||||
httpClientOverride = func(url string) (int, error) { return 200, nil }
|
||||
|
||||
// Provide a mock Traefik config so performCutover can read it.
|
||||
cfs := newMockCutoverFS()
|
||||
cfs.files["/etc/traefik/traefik.yml"] = []byte("entrypoint: :443\n")
|
||||
cutoverFSOverride = cfs
|
||||
t.Cleanup(func() { cutoverFSOverride = nil })
|
||||
|
||||
rootCmd.SetArgs([]string{"upgrade", "--to", "v0.11.0", "--force"})
|
||||
if err := rootCmd.Execute(); err != nil {
|
||||
t.Fatalf("upgrade with cutover: %v", err)
|
||||
@@ -180,6 +258,12 @@ func TestUpgradeCutoverRollback(t *testing.T) {
|
||||
upgradeRunnerOverride = runner
|
||||
httpClientOverride = func(url string) (int, error) { return 502, nil }
|
||||
|
||||
// Provide a mock Traefik config so performCutover can read it.
|
||||
cfs := newMockCutoverFS()
|
||||
cfs.files["/etc/traefik/traefik.yml"] = []byte("entrypoint: :443\n")
|
||||
cutoverFSOverride = cfs
|
||||
t.Cleanup(func() { cutoverFSOverride = nil })
|
||||
|
||||
rootCmd.SetArgs([]string{"upgrade", "--to", "v0.11.0"})
|
||||
err := rootCmd.Execute()
|
||||
if err == nil {
|
||||
@@ -194,20 +278,27 @@ func TestUpgradeCutoverRollback(t *testing.T) {
|
||||
t.Errorf("output should mention rollback: %s", out)
|
||||
}
|
||||
|
||||
// Verify rollback: the traefik.yml content should be restored to
|
||||
// :443 (the backup was renamed back over the modified file).
|
||||
restored, ok := cfs.files["/etc/traefik/traefik.yml"]
|
||||
if !ok {
|
||||
t.Fatal("rollback: traefik.yml missing after rollback")
|
||||
}
|
||||
if !strings.Contains(string(restored), ":443") {
|
||||
t.Errorf("rollback: traefik.yml not restored to :443, got: %s", string(restored))
|
||||
}
|
||||
if strings.Contains(string(restored), "127.0.0.1:8443") {
|
||||
t.Errorf("rollback: traefik.yml still has 127.0.0.1:8443 after rollback: %s", string(restored))
|
||||
}
|
||||
|
||||
foundRollback := false
|
||||
for _, call := range runner.calls {
|
||||
if call.name == "sed" && len(call.args) >= 2 {
|
||||
joined := strings.Join(call.args, " ")
|
||||
if strings.Contains(joined, "127.0.0.1:8443") && strings.Contains(joined, ":443") {
|
||||
foundRollback = true
|
||||
}
|
||||
}
|
||||
if call.name == "nft" && len(call.args) >= 2 && call.args[0] == "delete" {
|
||||
foundRollback = true
|
||||
}
|
||||
}
|
||||
if !foundRollback {
|
||||
t.Errorf("rollback commands not detected (calls: %v)", runner.calls)
|
||||
t.Errorf("rollback nft delete command not detected (calls: %v)", runner.calls)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -227,6 +318,12 @@ func TestUpgradeCutoverForceSkipsVerification(t *testing.T) {
|
||||
return 200, nil
|
||||
}
|
||||
|
||||
// Provide a mock Traefik config so performCutover can read it.
|
||||
cfs := newMockCutoverFS()
|
||||
cfs.files["/etc/traefik/traefik.yml"] = []byte("entrypoint: :443\n")
|
||||
cutoverFSOverride = cfs
|
||||
t.Cleanup(func() { cutoverFSOverride = nil })
|
||||
|
||||
rootCmd.SetArgs([]string{"upgrade", "--to", "v0.11.0", "--force"})
|
||||
if err := rootCmd.Execute(); err != nil {
|
||||
t.Fatalf("upgrade with --force: %v", err)
|
||||
@@ -339,3 +436,211 @@ func TestUpgradeFullMigration(t *testing.T) {
|
||||
t.Errorf("install.sh was not invoked (calls: %v)", runner.calls)
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
// TestCutoverBackupRestoreOnFailure verifies that when the cutover
|
||||
// verification fails, the Traefik config is restored from the backup
|
||||
// file (REQ-158, P09 T10). This is a unit-level test that calls
|
||||
// performCutover directly with a mock FS.
|
||||
func TestCutoverBackupRestoreOnFailure(t *testing.T) {
|
||||
// Set up a mock FS with a Traefik config containing :443.
|
||||
cfs := newMockCutoverFS()
|
||||
original := []byte("entrypoint:\n - :443\n")
|
||||
cfs.files["/etc/traefik/traefik.yml"] = original
|
||||
cutoverFSOverride = cfs
|
||||
t.Cleanup(func() { cutoverFSOverride = nil })
|
||||
|
||||
// Mock runner that succeeds for systemctl restart.
|
||||
runner := &mockUpgradeRunner{
|
||||
outputs: make(map[string][]byte),
|
||||
}
|
||||
// Mock HTTP check returns 502 (failure).
|
||||
prevHTTP := httpClientOverride
|
||||
httpClientOverride = func(url string) (int, error) { return 502, nil }
|
||||
t.Cleanup(func() { httpClientOverride = prevHTTP })
|
||||
|
||||
var buf bytes.Buffer
|
||||
ok, err := performCutover(context.Background(), runner, &buf, false)
|
||||
if err != nil {
|
||||
t.Fatalf("performCutover: %v", err)
|
||||
}
|
||||
if ok {
|
||||
t.Fatal("expected cutover to fail (ok=false)")
|
||||
}
|
||||
|
||||
// Verify the Traefik config was restored from backup.
|
||||
restored, exists := cfs.files["/etc/traefik/traefik.yml"]
|
||||
if !exists {
|
||||
t.Fatal("traefik.yml missing after rollback")
|
||||
}
|
||||
if string(restored) != string(original) {
|
||||
t.Errorf("traefik.yml not restored to original, got: %s", string(restored))
|
||||
}
|
||||
// Verify 127.0.0.1:8443 is NOT in the restored file.
|
||||
if strings.Contains(string(restored), "127.0.0.1:8443") {
|
||||
t.Errorf("traefik.yml still has 127.0.0.1:8443 after rollback: %s", string(restored))
|
||||
}
|
||||
// The backup file should have been consumed by rollbackCutover's rename.
|
||||
if _, bakExists := cfs.files["/etc/traefik/traefik.yml.bak"]; bakExists {
|
||||
t.Error("backup file still exists after rollback (should have been renamed)")
|
||||
}
|
||||
}
|
||||
|
||||
// TestCutoverAtomicRenameSuccess verifies that the cutover writes the
|
||||
// new config via atomic rename (temp file → original) and cleans up
|
||||
// the backup on success (REQ-158, P09 T10).
|
||||
func TestCutoverAtomicRenameSuccess(t *testing.T) {
|
||||
cfs := newMockCutoverFS()
|
||||
original := []byte("entrypoint:\n - :443\n")
|
||||
cfs.files["/etc/traefik/traefik.yml"] = original
|
||||
cutoverFSOverride = cfs
|
||||
t.Cleanup(func() { cutoverFSOverride = nil })
|
||||
|
||||
runner := &mockUpgradeRunner{
|
||||
outputs: make(map[string][]byte),
|
||||
}
|
||||
prevHTTP := httpClientOverride
|
||||
httpClientOverride = func(url string) (int, error) { return 200, nil }
|
||||
t.Cleanup(func() { httpClientOverride = prevHTTP })
|
||||
|
||||
var buf bytes.Buffer
|
||||
ok, err := performCutover(context.Background(), runner, &buf, false)
|
||||
if err != nil {
|
||||
t.Fatalf("performCutover: %v", err)
|
||||
}
|
||||
if !ok {
|
||||
t.Fatal("expected cutover to succeed (ok=true)")
|
||||
}
|
||||
|
||||
// Verify the config was updated to 127.0.0.1:8443.
|
||||
updated, exists := cfs.files["/etc/traefik/traefik.yml"]
|
||||
if !exists {
|
||||
t.Fatal("traefik.yml missing after cutover")
|
||||
}
|
||||
if !strings.Contains(string(updated), "127.0.0.1:8443") {
|
||||
t.Errorf("traefik.yml should have 127.0.0.1:8443, got: %s", string(updated))
|
||||
}
|
||||
if strings.Contains(string(updated), ":443\n") && !strings.Contains(string(updated), "127.0.0.1:8443") {
|
||||
t.Errorf("traefik.yml should not have bare :443 anymore, got: %s", string(updated))
|
||||
}
|
||||
// The temp file should not exist.
|
||||
if _, tmpExists := cfs.files["/etc/traefik/traefik.yml.tmp"]; tmpExists {
|
||||
t.Error("temp file still exists after atomic rename")
|
||||
}
|
||||
// The backup should have been cleaned up on success.
|
||||
if _, bakExists := cfs.files["/etc/traefik/traefik.yml.bak"]; bakExists {
|
||||
t.Error("backup file still exists after successful cutover (should be cleaned up)")
|
||||
}
|
||||
}
|
||||
|
||||
// TestCutoverBackupCreated verifies that a backup file is created
|
||||
// before the cutover edits the config (REQ-158, P09 T10). Uses a
|
||||
// custom mock FS that records the sequence of operations so we can
|
||||
// assert the backup was written before the temp file.
|
||||
func TestCutoverBackupCreated(t *testing.T) {
|
||||
// Use a recording mock FS that fails on the rename step so the
|
||||
// backup write is observable before the rollback consumes it.
|
||||
cfs := newMockCutoverFS()
|
||||
original := []byte("entrypoint:\n - :443\n")
|
||||
cfs.files["/etc/traefik/traefik.yml"] = original
|
||||
// Track write order via a custom FS that records operations.
|
||||
var writeOrder []string
|
||||
recordingCFS := &recordingCutoverFS{
|
||||
inner: cfs,
|
||||
writeOrder: &writeOrder,
|
||||
}
|
||||
// Make the rename of the temp file fail so the cutover aborts.
|
||||
cfs.errs["rename:/etc/traefik/traefik.yml.tmp"] = fmt.Errorf("rename failed")
|
||||
cutoverFSOverride = recordingCFS
|
||||
t.Cleanup(func() { cutoverFSOverride = nil })
|
||||
|
||||
runner := &mockUpgradeRunner{
|
||||
outputs: make(map[string][]byte),
|
||||
}
|
||||
|
||||
var buf bytes.Buffer
|
||||
_, err := performCutover(context.Background(), runner, &buf, false)
|
||||
if err == nil {
|
||||
t.Fatal("expected error from failed rename")
|
||||
}
|
||||
|
||||
// Verify the backup was written BEFORE the temp file.
|
||||
// writeOrder records WriteFile calls in order.
|
||||
bakIdx := -1
|
||||
tmpIdx := -1
|
||||
for i, p := range writeOrder {
|
||||
if p == "/etc/traefik/traefik.yml.bak" {
|
||||
bakIdx = i
|
||||
}
|
||||
if p == "/etc/traefik/traefik.yml.tmp" {
|
||||
tmpIdx = i
|
||||
}
|
||||
}
|
||||
if bakIdx == -1 {
|
||||
t.Fatal("backup file was not written before cutover")
|
||||
}
|
||||
if tmpIdx == -1 {
|
||||
t.Fatal("temp file was not written")
|
||||
}
|
||||
if bakIdx > tmpIdx {
|
||||
t.Errorf("backup written after temp file (bakIdx=%d, tmpIdx=%d) — backup should come first", bakIdx, tmpIdx)
|
||||
}
|
||||
// The original should have been restored from backup on failure.
|
||||
restored, exists := cfs.files["/etc/traefik/traefik.yml"]
|
||||
if !exists {
|
||||
t.Fatal("traefik.yml missing after failed rename + restore")
|
||||
}
|
||||
if string(restored) != string(original) {
|
||||
t.Errorf("traefik.yml not restored to original after failed rename, got: %s", string(restored))
|
||||
}
|
||||
}
|
||||
|
||||
// recordingCutoverFS wraps a cutoverFS and records WriteFile call
|
||||
// paths so tests can assert the order of operations (REQ-158, P09 T10).
|
||||
type recordingCutoverFS struct {
|
||||
inner cutoverFS
|
||||
writeOrder *[]string
|
||||
}
|
||||
|
||||
func (r *recordingCutoverFS) ReadFile(path string) ([]byte, error) {
|
||||
return r.inner.ReadFile(path)
|
||||
}
|
||||
func (r *recordingCutoverFS) WriteFile(path string, content []byte, mode os.FileMode) error {
|
||||
*r.writeOrder = append(*r.writeOrder, path)
|
||||
return r.inner.WriteFile(path, content, mode)
|
||||
}
|
||||
func (r *recordingCutoverFS) Rename(old, new string) error {
|
||||
return r.inner.Rename(old, new)
|
||||
}
|
||||
func (r *recordingCutoverFS) Remove(path string) error {
|
||||
return r.inner.Remove(path)
|
||||
}
|
||||
func (r *recordingCutoverFS) Stat(path string) (os.FileInfo, error) {
|
||||
return r.inner.Stat(path)
|
||||
}
|
||||
|
||||
// TestCutoverNoSedDirectly verifies that the cutover does NOT use
|
||||
// `sed -i` (the old unsafe approach). The mock runner records all
|
||||
// calls; none should be `sed` (REQ-158, P09 T5).
|
||||
func TestCutoverNoSedDirectly(t *testing.T) {
|
||||
cfs := newMockCutoverFS()
|
||||
cfs.files["/etc/traefik/traefik.yml"] = []byte("entrypoint:\n - :443\n")
|
||||
cutoverFSOverride = cfs
|
||||
t.Cleanup(func() { cutoverFSOverride = nil })
|
||||
|
||||
runner := &mockUpgradeRunner{
|
||||
outputs: make(map[string][]byte),
|
||||
}
|
||||
prevHTTP := httpClientOverride
|
||||
httpClientOverride = func(url string) (int, error) { return 200, nil }
|
||||
t.Cleanup(func() { httpClientOverride = prevHTTP })
|
||||
|
||||
var buf bytes.Buffer
|
||||
_, _ = performCutover(context.Background(), runner, &buf, false)
|
||||
|
||||
for _, call := range runner.calls {
|
||||
if call.name == "sed" {
|
||||
t.Errorf("cutover should not use 'sed' (uses atomic rename now), found call: %s %v", call.name, call.args)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -98,24 +98,39 @@ type TaskSpec struct {
|
||||
}
|
||||
|
||||
func (e *Executor) Run(ctx context.Context, job *model.Job, specs []TaskSpec) error {
|
||||
e.mu.Lock()
|
||||
defer e.mu.Unlock()
|
||||
// REQ-156 / P07 T6: the mutex previously guarded the ENTIRE job
|
||||
// (insert + status transitions + task execution + wait). That
|
||||
// serialized unrelated jobs against each other and held the lock
|
||||
// across long-running child processes, blocking concurrent
|
||||
// Submit/Status/Run callers. The mutex is now scoped ONLY to the
|
||||
// DB inserts/updates (the part that must be serialized against
|
||||
// the single-writer SQLite connection pool — see store.Open
|
||||
// SetMaxOpenConns(1)). The task goroutines spawned below do not
|
||||
// hold e.mu; they share the per-job failure counter via a local
|
||||
// sync.Mutex.
|
||||
|
||||
// Insert the job first so tasks can reference it via foreign key.
|
||||
// Insert the job + flip to Running under the lock (serializes
|
||||
// the DB writes; the underlying SQLite busy_timeout(5000) +
|
||||
// SetMaxOpenConns(1) handles contention).
|
||||
e.mu.Lock()
|
||||
if err := e.jobs.Insert(ctx, job); err != nil {
|
||||
e.mu.Unlock()
|
||||
return err
|
||||
}
|
||||
if err := e.jobs.UpdateStatus(ctx, job.ID, model.JobStatusRunning, 0); err != nil {
|
||||
e.mu.Unlock()
|
||||
return err
|
||||
}
|
||||
e.mu.Unlock()
|
||||
|
||||
// Task execution runs WITHOUT e.mu — concurrent jobs (and
|
||||
// concurrent Submit/Status callers) are no longer blocked by a
|
||||
// long-running child process.
|
||||
var (
|
||||
wg sync.WaitGroup
|
||||
failedCount int
|
||||
exitCode int
|
||||
mu sync.Mutex
|
||||
)
|
||||
|
||||
for _, ts := range specs {
|
||||
wg.Add(1)
|
||||
go func(ts TaskSpec) {
|
||||
@@ -133,14 +148,16 @@ func (e *Executor) Run(ctx context.Context, job *model.Job, specs []TaskSpec) er
|
||||
}
|
||||
wg.Wait()
|
||||
|
||||
// Final status transition under the lock (the DB write is the
|
||||
// only thing that needs serialization).
|
||||
e.mu.Lock()
|
||||
defer e.mu.Unlock()
|
||||
if failedCount > 0 {
|
||||
exitCode = 1
|
||||
if err := e.jobs.UpdateStatus(ctx, job.ID, model.JobStatusFailed, exitCode); err != nil {
|
||||
if err := e.jobs.UpdateStatus(ctx, job.ID, model.JobStatusFailed, 1); err != nil {
|
||||
return err
|
||||
}
|
||||
return fmt.Errorf("%d/%d tasks failed", failedCount, len(specs))
|
||||
}
|
||||
|
||||
if err := e.jobs.UpdateStatus(ctx, job.ID, model.JobStatusComplete, 0); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
+18
-12
@@ -29,6 +29,8 @@ import (
|
||||
|
||||
"github.com/coreos/go-oidc/v3/oidc"
|
||||
"golang.org/x/oauth2"
|
||||
|
||||
"git.cloudinit.dev/coreci/orca/internal/security"
|
||||
)
|
||||
|
||||
// OIDCConfig holds the OIDC client configuration. It is loaded from
|
||||
@@ -113,7 +115,13 @@ func SaveCredentials(c *Credentials) error {
|
||||
if err != nil {
|
||||
return fmt.Errorf("oidc: marshal: %w", err)
|
||||
}
|
||||
return writeAtomic0600(path, data)
|
||||
// REQ-156 / P07 T9: use the canonical security.WriteAtomic (temp
|
||||
// + chmod + fsync + rename) instead of the local writeAtomic0600
|
||||
// (which did temp + chmod + rename with NO fsync - a crash before
|
||||
// rename could leave a partially-written tmp file that rename
|
||||
// would then promote, or the rename could land before the data
|
||||
// reached durable storage).
|
||||
return security.WriteAtomic(path, 0o600, data)
|
||||
}
|
||||
|
||||
// ClearCredentials removes the stored credentials (logout).
|
||||
@@ -128,16 +136,6 @@ func ClearCredentials() error {
|
||||
return nil
|
||||
}
|
||||
|
||||
// writeAtomic0600 writes data to path atomically at mode 0600
|
||||
// (temp + chmod + rename).
|
||||
func writeAtomic0600(path string, data []byte) error {
|
||||
tmp := path + ".tmp"
|
||||
if err := os.WriteFile(tmp, data, 0o600); err != nil {
|
||||
return fmt.Errorf("oidc: write tmp: %w", err)
|
||||
}
|
||||
return os.Rename(tmp, path)
|
||||
}
|
||||
|
||||
// OIDCClient wraps the OIDC provider + oauth2 config for the auth flow.
|
||||
type OIDCClient struct {
|
||||
provider *oidc.Provider
|
||||
@@ -237,7 +235,15 @@ func (c *OIDCClient) Login(ctx context.Context, openBrowser func(string) error)
|
||||
err error
|
||||
}
|
||||
resultCh := make(chan result, 1)
|
||||
srv := &http.Server{}
|
||||
// REQ-157 / P08 T8: set ReadHeaderTimeout so a slowloris-style
|
||||
// peer cannot hold the callback server open indefinitely. The
|
||||
// callback is short-lived (one request then Shutdown), but the
|
||||
// default zero ReadHeaderTimeout means an attacker who reaches the
|
||||
// loopback port during the brief auth window could stall the
|
||||
// handshake. 5s is generous for a loopback redirect.
|
||||
srv := &http.Server{
|
||||
ReadHeaderTimeout: 5 * time.Second,
|
||||
}
|
||||
srv.Handler = http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
if r.URL.Path != "/callback" {
|
||||
http.NotFound(w, r)
|
||||
|
||||
@@ -0,0 +1,216 @@
|
||||
// Package linux implements the SSH-based bootstrap of a generic Linux
|
||||
// host (Ubuntu/Debian/Alpine) as an orca worker node (REQ-161, P12).
|
||||
//
|
||||
// The bootstrap sequence (run via `orca node join --type linux`):
|
||||
// 1. Generate or load the orca SSH keypair (Ed25519, D-037)
|
||||
// 2. SSH dial with key auth + TOFU host-key capture (D-035)
|
||||
// 3. Deploy the orca pubkey to ~orca/.ssh/authorized_keys
|
||||
// 4. Create the `orca` Linux system user (nologin shell)
|
||||
// 5. Create the drift-events directory (~orca/drift-events)
|
||||
// 6. Return the node metadata for the caller to persist
|
||||
//
|
||||
// Unlike Proxmox bootstrap, there is NO PVE role, NO sudoers file, and
|
||||
// NO PVE user — this is a plain Linux worker. Authentication is
|
||||
// key-based (R-021): the orca SSH key is used for the initial SSH auth
|
||||
// and pubkey deployment; subsequent orca→worker access uses the same
|
||||
// key.
|
||||
//
|
||||
// All steps are idempotent: re-running the bootstrap on an
|
||||
// already-configured host is a no-op.
|
||||
package linux
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"fmt"
|
||||
"log/slog"
|
||||
"net"
|
||||
"os"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"golang.org/x/crypto/ssh"
|
||||
"golang.org/x/crypto/ssh/knownhosts"
|
||||
|
||||
"git.cloudinit.dev/coreci/orca/internal/certpaths"
|
||||
"git.cloudinit.dev/coreci/orca/internal/security"
|
||||
)
|
||||
|
||||
// DefaultSSHUser is the default SSH username for the initial connection.
|
||||
const DefaultSSHUser = "root"
|
||||
|
||||
// DefaultOrcaUser is the default Linux system user created on the worker.
|
||||
const DefaultOrcaUser = "orca"
|
||||
|
||||
// DefaultSSHPort is the default SSH port.
|
||||
const DefaultSSHPort = 22
|
||||
|
||||
// Options configures a Linux worker bootstrap run.
|
||||
type Options struct {
|
||||
Host string
|
||||
SSHUser string
|
||||
SSHKeyPath string
|
||||
OrcaUser string
|
||||
SSHPort int
|
||||
HostKeyFingerprint string
|
||||
Logger *slog.Logger
|
||||
}
|
||||
|
||||
// Result is the outcome of a successful bootstrap.
|
||||
type Result struct {
|
||||
NodeName string
|
||||
NodeAddress string
|
||||
HostKeyFingerprint string
|
||||
}
|
||||
|
||||
// BootstrapLinux runs the full SSH bootstrap sequence on a remote
|
||||
// generic Linux host. Returns the node metadata for the caller to
|
||||
// persist to the registry.
|
||||
func BootstrapLinux(ctx context.Context, opts Options) (*Result, error) {
|
||||
if opts.Host == "" {
|
||||
return nil, fmt.Errorf("linux bootstrap: --host is required")
|
||||
}
|
||||
if opts.SSHKeyPath == "" {
|
||||
return nil, fmt.Errorf("linux bootstrap: --ssh-key is required (R-021: no passwords; use --ssh-key or pre-stage the orca key)")
|
||||
}
|
||||
if opts.SSHUser == "" {
|
||||
opts.SSHUser = DefaultSSHUser
|
||||
}
|
||||
if opts.OrcaUser == "" {
|
||||
opts.OrcaUser = DefaultOrcaUser
|
||||
}
|
||||
if opts.SSHPort == 0 {
|
||||
opts.SSHPort = DefaultSSHPort
|
||||
}
|
||||
if opts.Logger == nil {
|
||||
opts.Logger = slog.Default()
|
||||
}
|
||||
|
||||
// Step 1: Load the orca SSH keypair.
|
||||
privKey, err := os.ReadFile(opts.SSHKeyPath)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("linux bootstrap: read SSH key: %w", err)
|
||||
}
|
||||
signer, err := ssh.ParsePrivateKey(privKey)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("linux bootstrap: parse SSH key: %w", err)
|
||||
}
|
||||
pubKey, err := os.ReadFile(certpaths.SSHPubPath())
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("linux bootstrap: read orca pubkey: %w", err)
|
||||
}
|
||||
pubKeyLine := strings.TrimSpace(string(pubKey))
|
||||
|
||||
// Step 2: SSH dial with key auth + TOFU host-key capture.
|
||||
sshAddr := net.JoinHostPort(opts.Host, fmt.Sprintf("%d", opts.SSHPort))
|
||||
var capturedHostKey ssh.PublicKey
|
||||
var hostKeyCallback ssh.HostKeyCallback
|
||||
if opts.HostKeyFingerprint != "" {
|
||||
hkcb, err := pinnedHostKeyCallback(opts.HostKeyFingerprint, &capturedHostKey)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("linux bootstrap: parse host key fingerprint: %w", err)
|
||||
}
|
||||
hostKeyCallback = hkcb
|
||||
} else {
|
||||
hkcb, err := knownhosts.New(certpaths.KnownHostsPath())
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("linux bootstrap: known_hosts: %w", err)
|
||||
}
|
||||
hostKeyCallback = ssh.HostKeyCallback(func(hostname string, remote net.Addr, key ssh.PublicKey) error {
|
||||
err := hkcb(hostname, remote, key)
|
||||
if err == nil {
|
||||
capturedHostKey = key
|
||||
}
|
||||
return err
|
||||
})
|
||||
}
|
||||
|
||||
sshConfig := &ssh.ClientConfig{
|
||||
User: opts.SSHUser,
|
||||
Auth: []ssh.AuthMethod{ssh.PublicKeys(signer)},
|
||||
HostKeyCallback: hostKeyCallback,
|
||||
Timeout: 30 * time.Second,
|
||||
}
|
||||
|
||||
opts.Logger.Info("linux bootstrap: dialing", "addr", sshAddr, "user", opts.SSHUser)
|
||||
client, err := ssh.Dial("tcp", sshAddr, sshConfig)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("linux bootstrap: SSH dial %s: %w", sshAddr, err)
|
||||
}
|
||||
defer client.Close()
|
||||
|
||||
// Step 3: Deploy the orca pubkey to authorized_keys.
|
||||
if err := sshExec(client, fmt.Sprintf(
|
||||
"mkdir -p ~%s/.ssh && grep -qF '%s' ~%s/.ssh/authorized_keys 2>/dev/null || echo '%s' >> ~%s/.ssh/authorized_keys && chmod 700 ~%s/.ssh && chmod 600 ~%s/.ssh/authorized_keys",
|
||||
opts.OrcaUser, pubKeyLine, opts.OrcaUser, pubKeyLine, opts.OrcaUser, opts.OrcaUser, opts.OrcaUser,
|
||||
)); err != nil {
|
||||
return nil, fmt.Errorf("linux bootstrap: deploy pubkey: %w", err)
|
||||
}
|
||||
opts.Logger.Info("linux bootstrap: pubkey deployed", "user", opts.OrcaUser)
|
||||
|
||||
// Step 4: Create the orca system user (nologin shell).
|
||||
if err := sshExec(client, fmt.Sprintf(
|
||||
"id -u %s 2>/dev/null || useradd -r -s /usr/sbin/nologin -d /home/%s -m %s",
|
||||
opts.OrcaUser, opts.OrcaUser, opts.OrcaUser,
|
||||
)); err != nil {
|
||||
return nil, fmt.Errorf("linux bootstrap: create user: %w", err)
|
||||
}
|
||||
opts.Logger.Info("linux bootstrap: user created", "user", opts.OrcaUser)
|
||||
|
||||
// Step 5: Create the drift-events directory.
|
||||
if err := sshExec(client, fmt.Sprintf(
|
||||
"mkdir -p ~%s/drift-events && chown %s:%s ~%s/drift-events",
|
||||
opts.OrcaUser, opts.OrcaUser, opts.OrcaUser, opts.OrcaUser,
|
||||
)); err != nil {
|
||||
return nil, fmt.Errorf("linux bootstrap: create drift-events dir: %w", err)
|
||||
}
|
||||
opts.Logger.Info("linux bootstrap: drift-events dir created", "user", opts.OrcaUser)
|
||||
|
||||
// Step 6: Return node metadata.
|
||||
hostKeyFP := ""
|
||||
if capturedHostKey != nil {
|
||||
hostKeyFP = ssh.FingerprintSHA256(capturedHostKey)
|
||||
}
|
||||
|
||||
return &Result{
|
||||
NodeName: opts.Host,
|
||||
NodeAddress: fmt.Sprintf("%s:8443", opts.Host),
|
||||
HostKeyFingerprint: hostKeyFP,
|
||||
}, nil
|
||||
}
|
||||
|
||||
// sshExec runs a command on the remote host and returns an error if
|
||||
// the exit code is non-zero.
|
||||
func sshExec(client *ssh.Client, cmd string) error {
|
||||
session, err := client.NewSession()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer session.Close()
|
||||
var stderr bytes.Buffer
|
||||
session.Stderr = &stderr
|
||||
if err := session.Run(cmd); err != nil {
|
||||
return fmt.Errorf("%w: %s", err, strings.TrimSpace(stderr.String()))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// pinnedHostKeyCallback returns a host key callback that pins to the
|
||||
// expected fingerprint.
|
||||
func pinnedHostKeyCallback(expectedSHA256Base64 string, capturedKey *ssh.PublicKey) (ssh.HostKeyCallback, error) {
|
||||
if expectedSHA256Base64 == "" {
|
||||
return nil, fmt.Errorf("empty fingerprint")
|
||||
}
|
||||
cb := ssh.HostKeyCallback(func(hostname string, remote net.Addr, key ssh.PublicKey) error {
|
||||
got := ssh.FingerprintSHA256(key)
|
||||
if got != expectedSHA256Base64 {
|
||||
return fmt.Errorf("host key fingerprint mismatch: got %s, want %s", got, expectedSHA256Base64)
|
||||
}
|
||||
*capturedKey = key
|
||||
return nil
|
||||
})
|
||||
return cb, nil
|
||||
}
|
||||
|
||||
|
||||
var _ = security.WriteAtomic
|
||||
@@ -3,7 +3,7 @@
|
||||
//
|
||||
// The bootstrap sequence (run via `orca node join --type proxmox`):
|
||||
// 1. Generate or load the orca SSH keypair (Ed25519, D-037)
|
||||
// 2. SSH dial with password auth + TOFU host-key capture (D-035)
|
||||
// 2. SSH dial with key auth + TOFU host-key capture (D-035)
|
||||
// 3. Deploy the orca pubkey to ~orca/.ssh/authorized_keys
|
||||
// 4. Create the `orca` Linux system user (config-overridable name)
|
||||
// 5. Create the OrcaOperator PVE role with least-privilege privileges
|
||||
@@ -16,9 +16,9 @@
|
||||
// 10. Return the node metadata for the caller to persist
|
||||
//
|
||||
// All steps are idempotent (D-036): re-running the bootstrap on an
|
||||
// already-configured host is a no-op. The password is never persisted
|
||||
// (D-031) — it is used only for the initial SSH auth and pubkey
|
||||
// deployment; subsequent orca→Proxmox access uses the deployed SSH key.
|
||||
// already-configured host is a no-op. Authentication is key-based (R-021)
|
||||
// (D-031): the orca SSH key is used for the initial SSH auth and
|
||||
// pubkey deployment; subsequent orca→Proxmox access uses the same key.
|
||||
package proxmox
|
||||
|
||||
import (
|
||||
@@ -143,14 +143,14 @@ func BootstrapProxmox(ctx context.Context, opts Options) (*Result, error) {
|
||||
return nil, fmt.Errorf("ssh key: %w", err)
|
||||
}
|
||||
|
||||
// Step 2: SSH dial with password auth + host-key verification (D-035,
|
||||
// Step 2: SSH dial with key auth + host-key verification (D-035,
|
||||
// REQ-058). When opts.HostKeyFingerprint is set (D-044), use a pinned
|
||||
// callback that fails closed on mismatch (AD-028); otherwise use the
|
||||
// TOFU known_hosts capture callback (D-035). The TOFU wrapper fixes
|
||||
// the v0.6 ship-defect where knownhosts.New returned KeyError{Want:[]}
|
||||
// on first connect WITHOUT writing the captured key, so the first
|
||||
// `orca node join --type proxmox` always failed.
|
||||
sshAddr := fmt.Sprintf("%s:%d", opts.Host, opts.SSHPort)
|
||||
sshAddr := net.JoinHostPort(opts.Host, fmt.Sprintf("%d", opts.SSHPort))
|
||||
var capturedHostKey ssh.PublicKey
|
||||
var hostKeyCallback ssh.HostKeyCallback
|
||||
if opts.HostKeyFingerprint != "" {
|
||||
@@ -296,8 +296,29 @@ func pinnedHostKeyCallback(expectedSHA256Base64 string, capturedKey *ssh.PublicK
|
||||
//
|
||||
// Exported so the doctor proxmox probe (T02.9) can reuse the same
|
||||
// capture-fix wrapper for parity (GRILL condition #2).
|
||||
//
|
||||
// REQ-157 / P08 T4: TOFUHostKeyCallback now delegates to
|
||||
// TOFUHostKeyCallbackPath with the v0.8 flat layout
|
||||
// (certpaths.KnownHostsPath()). The path-accepting variant lets the
|
||||
// sshpush transport pass its stored known_hosts field (the v0.9
|
||||
// paths.KnownHostsPath() location) instead of always reading the v0.8
|
||||
// flat layout — fixing the bug where the dial() flock field was stored
|
||||
// but never read.
|
||||
func TOFUHostKeyCallback(addr string, capturedKey *ssh.PublicKey) (ssh.HostKeyCallback, error) {
|
||||
cb, err := knownhosts.New(certpaths.KnownHostsPath())
|
||||
return TOFUHostKeyCallbackPath(certpaths.KnownHostsPath(), addr, capturedKey)
|
||||
}
|
||||
|
||||
// TOFUHostKeyCallbackPath is the path-accepting variant. knownHostsPath
|
||||
// is the known_hosts file to verify against and capture new keys into;
|
||||
// it MUST be flock-protected on capture (security.Flock). When
|
||||
// knownHostsPath is empty, falls back to certpaths.KnownHostsPath()
|
||||
// (the v0.8 flat layout) for backward compatibility with callers that
|
||||
// relied on the implicit default.
|
||||
func TOFUHostKeyCallbackPath(knownHostsPath, addr string, capturedKey *ssh.PublicKey) (ssh.HostKeyCallback, error) {
|
||||
if knownHostsPath == "" {
|
||||
knownHostsPath = certpaths.KnownHostsPath()
|
||||
}
|
||||
cb, err := knownhosts.New(knownHostsPath)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
@@ -312,13 +333,12 @@ func TOFUHostKeyCallback(addr string, capturedKey *ssh.PublicKey) (ssh.HostKeyCa
|
||||
var keyErr *knownhosts.KeyError
|
||||
if errors.As(err, &keyErr) && len(keyErr.Want) == 0 {
|
||||
line := knownhosts.Line([]string{knownhosts.Normalize(addr)}, key)
|
||||
path := certpaths.KnownHostsPath()
|
||||
release, lockErr := security.Flock(path)
|
||||
release, lockErr := security.Flock(knownHostsPath)
|
||||
if lockErr != nil {
|
||||
return fmt.Errorf("tofu lock known_hosts: %w", lockErr)
|
||||
}
|
||||
defer release()
|
||||
existing, readErr := os.ReadFile(path)
|
||||
existing, readErr := os.ReadFile(knownHostsPath)
|
||||
if readErr != nil && !os.IsNotExist(readErr) {
|
||||
return fmt.Errorf("tofu read known_hosts: %w", readErr)
|
||||
}
|
||||
@@ -326,7 +346,7 @@ func TOFUHostKeyCallback(addr string, capturedKey *ssh.PublicKey) (ssh.HostKeyCa
|
||||
existing = append(existing, '\n')
|
||||
}
|
||||
updated := append(existing, []byte(line)...)
|
||||
if writeErr := security.WriteAtomic(path, 0o600, updated); writeErr != nil {
|
||||
if writeErr := security.WriteAtomic(knownHostsPath, 0o600, updated); writeErr != nil {
|
||||
return fmt.Errorf("tofu write known_hosts: %w", writeErr)
|
||||
}
|
||||
if capturedKey != nil {
|
||||
|
||||
@@ -0,0 +1,30 @@
|
||||
package proxmox
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"net"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// TestREQ157_IPv6JoinHostPort verifies that the proxmox SSH dial
|
||||
// address is correctly bracketed for IPv6 hosts (REQ-157 / P08 T5/T11).
|
||||
func TestREQ157_IPv6JoinHostPort(t *testing.T) {
|
||||
tests := []struct {
|
||||
host string
|
||||
port int
|
||||
want string
|
||||
}{
|
||||
{"192.168.1.1", 22, "192.168.1.1:22"},
|
||||
{"::1", 22, "[::1]:22"},
|
||||
{"fe80::1", 2222, "[fe80::1]:2222"},
|
||||
{"2001:db8::1", 22, "[2001:db8::1]:22"},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.host, func(t *testing.T) {
|
||||
got := net.JoinHostPort(tt.host, fmt.Sprintf("%d", tt.port))
|
||||
if got != tt.want {
|
||||
t.Errorf("JoinHostPort(%s, %d) = %q, want %q", tt.host, tt.port, got, tt.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -5,11 +5,13 @@ import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"math/rand"
|
||||
"net"
|
||||
"os"
|
||||
"strings"
|
||||
"sync"
|
||||
"syscall"
|
||||
"time"
|
||||
|
||||
"golang.org/x/crypto/ssh"
|
||||
@@ -54,12 +56,14 @@ type Transport struct {
|
||||
pool sync.Map
|
||||
// keyPath is the SSH private key path (Ed25519, D-037).
|
||||
keyPath string
|
||||
// knownHostsPath is the v0.9 known_hosts path (paths.KnownHostsPath()
|
||||
// = ClusterDir()/known_hosts). It is stored for the v0.10-P14 migration
|
||||
// when proxmox.TOFUHostKeyCallback will accept a path parameter; today
|
||||
// the callback reads certpaths.KnownHostsPath() (the v0.8 flat layout)
|
||||
// directly, so this field is not yet read by dial(). Tests set
|
||||
// $ORCA_HOME so certpaths.KnownHostsPath() resolves under the temp dir.
|
||||
// knownHostsPath is the known_hosts path passed to the TOFU
|
||||
// host-key callback (D-035). NewTransport sets it from
|
||||
// certpaths.KnownHostsPath() (v0.8 flat layout) by default; callers
|
||||
// that want the v0.9 paths.KnownHostsPath() location construct the
|
||||
// transport with that path explicitly. REQ-157 / P08 T4: this field
|
||||
// IS read by dial() (via proxmox.TOFUHostKeyCallbackPath) — the
|
||||
// earlier bug where the callback ignored it and read
|
||||
// certpaths.KnownHostsPath() directly is fixed.
|
||||
knownHostsPath string
|
||||
// user is the remote SSH user (default "orca", D-037).
|
||||
user string
|
||||
@@ -120,15 +124,14 @@ func (defaultSSHDialer) DialContext(ctx context.Context, network, addr string, c
|
||||
}
|
||||
|
||||
// NewTransport returns a Transport configured with the given SSH
|
||||
// private key path and known_hosts path. The known_hosts path is the v0.9
|
||||
// location (paths.KnownHostsPath); it is stored for the v0.10-P14
|
||||
// migration when the TOFU callback will accept a path parameter. Today
|
||||
// dial() delegates host-key verification to proxmox.TOFUHostKeyCallback,
|
||||
// which reads certpaths.KnownHostsPath() (the v0.8 flat layout under
|
||||
// $ORCA_HOME) directly — so callers must ensure $ORCA_HOME points at the
|
||||
// cluster root (the CLI sets this up). The remote user defaults to
|
||||
// "orca" (D-037); override with SetUser. The dialer defaults to the
|
||||
// real ssh.Dial-based dialer; tests call SetDialer to inject a mock.
|
||||
// private key path and known_hosts path. The known_hosts path is read
|
||||
// by dial() via proxmox.TOFUHostKeyCallbackPath (D-035, REQ-157/P08 T4):
|
||||
// the TOFU callback locks/captures against this path on first connect.
|
||||
// Callers typically pass certpaths.KnownHostsPath() (the v0.8 flat
|
||||
// layout under $ORCA_HOME) or paths.KnownHostsPath() (the v0.9
|
||||
// ClusterDir() location). The remote user defaults to "orca" (D-037);
|
||||
// override with SetUser. The dialer defaults to the real ssh.Dial-based
|
||||
// dialer; tests call SetDialer to inject a mock.
|
||||
func NewTransport(keyPath, knownHostsPath string) *Transport {
|
||||
return &Transport{
|
||||
keyPath: keyPath,
|
||||
@@ -193,7 +196,14 @@ func (t *Transport) dial(peer string) (*ssh.Client, error) {
|
||||
// Host-key verification reuses the v0.8 TOFU wrapper (D-035). The
|
||||
// known_hosts file is flock-protected inside the callback on
|
||||
// first-connect capture, so we do NOT re-lock here.
|
||||
cb, err := proxmox.TOFUHostKeyCallback(peer, nil)
|
||||
//
|
||||
// REQ-157 / P08 T4: use the stored knownHostsPath field (set via
|
||||
// NewTransport from certpaths.KnownHostsPath() / paths.KnownHostsPath())
|
||||
// instead of having the callback read certpaths.KnownHostsPath() (the
|
||||
// v0.8 flat layout) directly. This closes the bug where the flock
|
||||
// field was stored but never read by dial() — the TOFU callback now
|
||||
// locks/captures against the path the transport was constructed with.
|
||||
cb, err := proxmox.TOFUHostKeyCallbackPath(t.knownHostsPath, peer, nil)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("sshpush: host-key callback: %w", err)
|
||||
}
|
||||
@@ -391,7 +401,14 @@ func backoff(initial, max time.Duration, n int) time.Duration {
|
||||
}
|
||||
|
||||
// isTransient reports whether err looks like a transient failure worth
|
||||
// retrying (mirrors v0.8 transport.IsTransient, reimplemented here).
|
||||
// retrying (mirrors transport.IsTransient, reimplemented here so
|
||||
// internal/sshpush does not import internal/transport).
|
||||
//
|
||||
// REQ-157 / P08 T2: classification is TYPE-BASED, not substring-based.
|
||||
// The primary path is errors.Is against the sentinels (ErrTransient /
|
||||
// ErrPermanent) and against well-known syscall/net/io errors. The
|
||||
// substring fallback is retained ONLY for unwrapped errors from the
|
||||
// ssh.Dialer that do not implement the standard interfaces.
|
||||
func isTransient(err error) bool {
|
||||
if err == nil {
|
||||
return false
|
||||
@@ -402,6 +419,29 @@ func isTransient(err error) bool {
|
||||
if errors.Is(err, ErrPermanent) {
|
||||
return false
|
||||
}
|
||||
// Typed: a net.Error that is a timeout is transient; a net.OpError
|
||||
// whose Temporary() is true (ECONNREFUSED et al) is transient.
|
||||
var netErr net.Error
|
||||
if errors.As(err, &netErr) {
|
||||
if netErr.Timeout() {
|
||||
return true
|
||||
}
|
||||
return isTemporarySSH(netErr)
|
||||
}
|
||||
if errors.Is(err, syscall.ECONNREFUSED) ||
|
||||
errors.Is(err, syscall.ECONNRESET) ||
|
||||
errors.Is(err, syscall.ETIMEDOUT) ||
|
||||
errors.Is(err, syscall.EHOSTUNREACH) ||
|
||||
errors.Is(err, syscall.ENETUNREACH) {
|
||||
return true
|
||||
}
|
||||
if errors.Is(err, io.EOF) || errors.Is(err, io.ErrUnexpectedEOF) {
|
||||
return true
|
||||
}
|
||||
if errors.Is(err, context.DeadlineExceeded) {
|
||||
return true
|
||||
}
|
||||
// Substring fallback (defense-in-depth for unwrapped errors).
|
||||
s := err.Error()
|
||||
for _, sub := range []string{
|
||||
"connection refused", "i/o timeout", "EOF",
|
||||
@@ -415,6 +455,17 @@ func isTransient(err error) bool {
|
||||
return false
|
||||
}
|
||||
|
||||
// isTemporarySSH reports whether netErr implements the legacy
|
||||
// Temporary() bool method and it returns true. net.OpError.Temporary()
|
||||
// maps to the underlying errno's temporary classification.
|
||||
func isTemporarySSH(netErr net.Error) bool {
|
||||
type temporary interface{ Temporary() bool }
|
||||
if t, ok := netErr.(temporary); ok {
|
||||
return t.Temporary()
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// classifyDialErr converts a raw ssh.Dial error into a transport error
|
||||
// (transient vs permanent). Auth failures and host-key mismatches are
|
||||
// permanent; everything else is transient.
|
||||
|
||||
@@ -22,6 +22,11 @@ func Open(path string) (*sql.DB, error) {
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("open sqlite: %w", err)
|
||||
}
|
||||
// REQ-156 / P07 T1: SQLite is a single-writer database. Cap the
|
||||
// connection pool at 1 so concurrent goroutines serialize on the
|
||||
// busy_timeout(5000) above instead of racing for the WAL writer
|
||||
// lock and surfacing spurious SQLITE_BUSY errors to callers.
|
||||
db.SetMaxOpenConns(1)
|
||||
if err := db.Ping(); err != nil {
|
||||
_ = db.Close()
|
||||
return nil, fmt.Errorf("ping sqlite: %w", err)
|
||||
|
||||
@@ -17,18 +17,37 @@ var metricOrder = []string{
|
||||
"txns_applied_total",
|
||||
"txns_drifted_total",
|
||||
"drifts_remediated_total",
|
||||
"drifts_remediated_total",
|
||||
"peers_total",
|
||||
"nodes_total",
|
||||
"allocs_total",
|
||||
"orca_jobs_running",
|
||||
"orca_jobs_failed",
|
||||
"orca_jobs_complete",
|
||||
"orca_audit_chain_head",
|
||||
"orca_drift_events_total",
|
||||
"orca_ssh_errors_total",
|
||||
"orca_txn_apply_total",
|
||||
"orca_txn_rollback_total",
|
||||
"orca_acl_denials_total",
|
||||
}
|
||||
|
||||
var metricMeta = map[string]metricDef{
|
||||
"txns_applied_total": {"Total transactions applied", "counter"},
|
||||
"txns_drifted_total": {"Total transactions drifted", "counter"},
|
||||
"drifts_remediated_total": {"Total drifts remediated", "counter"},
|
||||
"peers_total": {"Current peer count", "gauge"},
|
||||
"nodes_total": {"Current node count", "gauge"},
|
||||
"allocs_total": {"Current allocation count", "gauge"},
|
||||
"txns_applied_total": {"Total transactions applied", "counter"},
|
||||
"txns_drifted_total": {"Total transactions drifted", "counter"},
|
||||
"drifts_remediated_total": {"Total drifts remediated", "counter"},
|
||||
"peers_total": {"Current peer count", "gauge"},
|
||||
"nodes_total": {"Current node count", "gauge"},
|
||||
"allocs_total": {"Current allocation count", "gauge"},
|
||||
"orca_jobs_running": {"Jobs currently running", "gauge"},
|
||||
"orca_jobs_failed": {"Jobs that failed", "gauge"},
|
||||
"orca_jobs_complete": {"Jobs completed successfully", "gauge"},
|
||||
"orca_audit_chain_head": {"Audit chain integrity (1=verified)", "gauge"},
|
||||
"orca_drift_events_total": {"Total drift events detected", "counter"},
|
||||
"orca_ssh_errors_total": {"Total SSH errors", "counter"},
|
||||
"orca_txn_apply_total": {"Total transaction applies", "counter"},
|
||||
"orca_txn_rollback_total": {"Total transaction rollbacks", "counter"},
|
||||
"orca_acl_denials_total": {"Total ACL denials (enforce mode)", "counter"},
|
||||
}
|
||||
|
||||
type Metrics struct {
|
||||
|
||||
+81
-23
@@ -8,7 +8,11 @@ package transport
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"io"
|
||||
"math/rand"
|
||||
"net"
|
||||
"strings"
|
||||
"syscall"
|
||||
"time"
|
||||
)
|
||||
|
||||
@@ -34,29 +38,6 @@ func DefaultRetryPolicy() RetryPolicy {
|
||||
return RetryPolicy{Initial: RetryInitial, Max: RetryMax, MaxAttempts: RetryMaxAttempts}
|
||||
}
|
||||
|
||||
// IsTransient reports whether err looks like a transient failure
|
||||
// worth retrying. We treat network errors, context-deadline-exceeded
|
||||
// (peer was slow but reachable), and a sentinel ErrTransient as
|
||||
// retryable; everything else (4xx, validation, auth) is permanent.
|
||||
func IsTransient(err error) bool {
|
||||
if err == nil {
|
||||
return false
|
||||
}
|
||||
if errors.Is(err, ErrTransient) {
|
||||
return true
|
||||
}
|
||||
// We avoid pulling net/error here to keep dependencies minimal;
|
||||
// the most common transient signature is the substring "connection
|
||||
// refused" or "i/o timeout". Tests assert these explicitly.
|
||||
s := err.Error()
|
||||
for _, sub := range []string{"connection refused", "i/o timeout", "EOF", "no such host", "connection reset"} {
|
||||
if contains(s, sub) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// ErrTransient is a sentinel callers can wrap to mark an error
|
||||
// retryable. ErrPermanent is the opposite.
|
||||
var (
|
||||
@@ -64,6 +45,83 @@ var (
|
||||
ErrPermanent = errors.New("permanent error")
|
||||
)
|
||||
|
||||
// IsTransient reports whether err looks like a transient failure
|
||||
// worth retrying. We treat network errors, context-deadline-exceeded
|
||||
// (peer was slow but reachable), and a sentinel ErrTransient as
|
||||
// retryable; everything else (4xx, validation, auth) is permanent.
|
||||
//
|
||||
// REQ-157 / P08 T1: classification is TYPE-BASED, not substring-based.
|
||||
// The primary path is errors.Is against the sentinels (ErrTransient /
|
||||
// ErrPermanent) and against well-known syscall/net/io errors. The
|
||||
// substring fallback is retained ONLY for unwrapped errors from
|
||||
// third-party dialers that do not implement the standard interfaces
|
||||
// (defense-in-depth); callers SHOULD wrap with ErrTransient instead.
|
||||
func IsTransient(err error) bool {
|
||||
if err == nil {
|
||||
return false
|
||||
}
|
||||
// Explicit sentinels win.
|
||||
if errors.Is(err, ErrTransient) {
|
||||
return true
|
||||
}
|
||||
if errors.Is(err, ErrPermanent) {
|
||||
return false
|
||||
}
|
||||
// Typed classification: a net.Error that is a timeout is transient.
|
||||
var netErr net.Error
|
||||
if errors.As(err, &netErr) {
|
||||
if netErr.Timeout() {
|
||||
return true
|
||||
}
|
||||
// net.OpError implements Temporary(); that maps to the
|
||||
// underlying errno's temporary classification (ECONNREFUSED et
|
||||
// al). We keep the check so a plain "dial tcp: connection
|
||||
// refused" classifies as transient.
|
||||
return isTemporary(netErr)
|
||||
}
|
||||
// Specific syscall errors that are universally retryable.
|
||||
if errors.Is(err, syscall.ECONNREFUSED) ||
|
||||
errors.Is(err, syscall.ECONNRESET) ||
|
||||
errors.Is(err, syscall.ETIMEDOUT) ||
|
||||
errors.Is(err, syscall.EHOSTUNREACH) ||
|
||||
errors.Is(err, syscall.ENETUNREACH) {
|
||||
return true
|
||||
}
|
||||
// io.EOF on a read from a half-closed peer is transient (the
|
||||
// dispatch HTTP/2 path can surface this mid-stream).
|
||||
if errors.Is(err, io.EOF) || errors.Is(err, io.ErrUnexpectedEOF) {
|
||||
return true
|
||||
}
|
||||
// context.DeadlineExceeded from a slow-but-reachable peer is
|
||||
// transient (the next attempt may succeed under a fresh deadline).
|
||||
if errors.Is(err, context.DeadlineExceeded) {
|
||||
return true
|
||||
}
|
||||
// Substring fallback (defense-in-depth for unwrapped errors).
|
||||
s := err.Error()
|
||||
for _, sub := range []string{
|
||||
"connection refused", "i/o timeout", "EOF",
|
||||
"no such host", "connection reset",
|
||||
"deadline exceeded", "temporarily unavailable",
|
||||
} {
|
||||
if strings.Contains(s, sub) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// isTemporary reports whether netErr implements the legacy Temporary()
|
||||
// bool method and it returns true. net.OpError.Temporary() maps to the
|
||||
// underlying errno's temporary classification (ECONNREFUSED et al).
|
||||
func isTemporary(netErr net.Error) bool {
|
||||
type temporary interface{ Temporary() bool }
|
||||
if t, ok := netErr.(temporary); ok {
|
||||
return t.Temporary()
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// RetryableFunc is the signature Retry calls. It returns the result
|
||||
// and an error. The bool indicates whether the call is idempotent
|
||||
// (true = safe to retry without an idempotency key).
|
||||
|
||||
@@ -0,0 +1,49 @@
|
||||
package transport
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"io"
|
||||
"net"
|
||||
"testing"
|
||||
"fmt"
|
||||
|
||||
)
|
||||
|
||||
// TestREQ157_TypedErrorClassification verifies that IsTransient uses
|
||||
// typed sentinels and standard interfaces, not substring matching
|
||||
// (REQ-157 / P08 T10).
|
||||
func TestREQ157_TypedErrorClassification(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
err error
|
||||
want bool
|
||||
}{
|
||||
{"nil", nil, false},
|
||||
{"ErrTransient", ErrTransient, true},
|
||||
{"wrapped ErrTransient", fmt.Errorf("dial: %w", ErrTransient), true},
|
||||
{"ErrPermanent", ErrPermanent, false},
|
||||
{"wrapped ErrPermanent", fmt.Errorf("auth: %w", ErrPermanent), false},
|
||||
{"net timeout", &net.OpError{Op: "dial", Net: "tcp", Err: &timeoutError{}}, true},
|
||||
{"context deadline", context.DeadlineExceeded, true},
|
||||
{"context canceled", context.Canceled, false},
|
||||
{"io EOF", io.EOF, true},
|
||||
{"plain error", errors.New("some permanent error"), false},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
got := IsTransient(tt.err)
|
||||
if got != tt.want {
|
||||
t.Errorf("IsTransient(%v) = %v, want %v", tt.err, got, tt.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
type timeoutError struct{}
|
||||
|
||||
func (timeoutError) Error() string { return "i/o timeout" }
|
||||
func (timeoutError) Timeout() bool { return true }
|
||||
func (timeoutError) Temporary() bool { return true }
|
||||
|
||||
var _ = fmt.Errorf
|
||||
@@ -15,6 +15,7 @@ import (
|
||||
"fmt"
|
||||
"net/http"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"github.com/go-webauthn/webauthn/protocol"
|
||||
@@ -80,26 +81,54 @@ type RegistrationSession struct {
|
||||
CreatedAt time.Time
|
||||
}
|
||||
|
||||
// sessionStore holds in-flight sessions (registration + login). In
|
||||
// production this would be a Redis/shared cache; for the bundled
|
||||
// single-lead Dex, an in-memory map with TTL is sufficient.
|
||||
// sessionStore holds in-flight sessions (registration). In production
|
||||
// this would be a Redis/shared cache; for the bundled single-lead
|
||||
// Dex, an in-memory map with TTL is sufficient.
|
||||
//
|
||||
// REQ-156 / P07 T10: the session maps are accessed from HTTP handler
|
||||
// goroutines (one goroutine per request) and were previously plain
|
||||
// maps with no synchronization. Concurrent BeginRegistration calls
|
||||
// for the same username would race on map writes (detected by go
|
||||
// test -race in T11). A sync.Mutex now guards all access.
|
||||
type sessionStore struct {
|
||||
mu sync.Mutex
|
||||
sessions map[string]*RegistrationSession
|
||||
}
|
||||
|
||||
// regSessions is the global in-flight registration session store.
|
||||
var regSessions = &sessionStore{sessions: make(map[string]*RegistrationSession)}
|
||||
|
||||
// loginSessionStore holds in-flight login sessions (T10). Same
|
||||
// mutex pattern as sessionStore.
|
||||
type loginSessionStore struct {
|
||||
mu sync.Mutex
|
||||
sessions map[string]*LoginSession
|
||||
}
|
||||
|
||||
// loginSessions is the global in-flight login session store.
|
||||
var loginSessions = &loginSessionStore{sessions: make(map[string]*LoginSession)}
|
||||
|
||||
// sessionTTL is the max time a registration/login session is valid.
|
||||
const sessionTTL = 5 * time.Minute
|
||||
|
||||
// cleanSessions removes expired sessions.
|
||||
// cleanSessions removes expired registration + login sessions.
|
||||
// Called under each store's lock by the Begin* handlers.
|
||||
func cleanSessions() {
|
||||
now := time.Now()
|
||||
for id, s := range regSessions.sessions {
|
||||
if now.Sub(s.CreatedAt) > sessionTTL {
|
||||
regSessions.mu.Lock()
|
||||
for id, sess := range regSessions.sessions {
|
||||
if now.Sub(sess.CreatedAt) > sessionTTL {
|
||||
delete(regSessions.sessions, id)
|
||||
}
|
||||
}
|
||||
regSessions.mu.Unlock()
|
||||
loginSessions.mu.Lock()
|
||||
for id, sess := range loginSessions.sessions {
|
||||
if now.Sub(sess.CreatedAt) > sessionTTL {
|
||||
delete(loginSessions.sessions, id)
|
||||
}
|
||||
}
|
||||
loginSessions.mu.Unlock()
|
||||
}
|
||||
|
||||
// requireAuth checks the request for an authenticated session. When
|
||||
@@ -165,11 +194,13 @@ func (c *Connector) BeginRegistration(w http.ResponseWriter, r *http.Request) {
|
||||
return
|
||||
}
|
||||
sessionID := base64.RawURLEncoding.EncodeToString(userID)
|
||||
regSessions.mu.Lock()
|
||||
regSessions.sessions[sessionID] = &RegistrationSession{
|
||||
UserID: username,
|
||||
Challenge: session,
|
||||
CreatedAt: time.Now(),
|
||||
}
|
||||
regSessions.mu.Unlock()
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
json.NewEncoder(w).Encode(options)
|
||||
}
|
||||
@@ -188,16 +219,20 @@ func (c *Connector) FinishRegistration(w http.ResponseWriter, r *http.Request) {
|
||||
return
|
||||
}
|
||||
sessionID := base64.RawURLEncoding.EncodeToString([]byte(username))
|
||||
regSessions.mu.Lock()
|
||||
session, ok := regSessions.sessions[sessionID]
|
||||
if !ok {
|
||||
regSessions.mu.Unlock()
|
||||
http.Error(w, "no registration session; call /register first", http.StatusBadRequest)
|
||||
return
|
||||
}
|
||||
if time.Since(session.CreatedAt) > sessionTTL {
|
||||
delete(regSessions.sessions, sessionID)
|
||||
regSessions.mu.Unlock()
|
||||
http.Error(w, "session expired", http.StatusBadRequest)
|
||||
return
|
||||
}
|
||||
regSessions.mu.Unlock()
|
||||
parsed, err := protocol.ParseCredentialCreationResponseBody(r.Body)
|
||||
if err != nil {
|
||||
http.Error(w, fmt.Sprintf("parse attestation: %v", err), http.StatusBadRequest)
|
||||
@@ -221,7 +256,9 @@ func (c *Connector) FinishRegistration(w http.ResponseWriter, r *http.Request) {
|
||||
http.Error(w, fmt.Sprintf("store credential: %v", err), http.StatusInternalServerError)
|
||||
return
|
||||
}
|
||||
regSessions.mu.Lock()
|
||||
delete(regSessions.sessions, sessionID)
|
||||
regSessions.mu.Unlock()
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
json.NewEncoder(w).Encode(map[string]string{"status": "registered", "user_id": username})
|
||||
}
|
||||
@@ -233,8 +270,6 @@ type LoginSession struct {
|
||||
CreatedAt time.Time
|
||||
}
|
||||
|
||||
var loginSessions = map[string]*LoginSession{}
|
||||
|
||||
// BeginLogin starts the WebAuthn login ceremony.
|
||||
// GET /orca/webauthn/login?username=<name>
|
||||
func (c *Connector) BeginLogin(w http.ResponseWriter, r *http.Request) {
|
||||
@@ -259,11 +294,13 @@ func (c *Connector) BeginLogin(w http.ResponseWriter, r *http.Request) {
|
||||
http.Error(w, fmt.Sprintf("begin login: %v", err), http.StatusInternalServerError)
|
||||
return
|
||||
}
|
||||
loginSessions[username] = &LoginSession{
|
||||
loginSessions.mu.Lock()
|
||||
loginSessions.sessions[username] = &LoginSession{
|
||||
UserID: username,
|
||||
Challenge: session,
|
||||
CreatedAt: time.Now(),
|
||||
}
|
||||
loginSessions.mu.Unlock()
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
json.NewEncoder(w).Encode(options)
|
||||
}
|
||||
@@ -276,16 +313,20 @@ func (c *Connector) FinishLogin(w http.ResponseWriter, r *http.Request) {
|
||||
http.Error(w, "username required", http.StatusBadRequest)
|
||||
return
|
||||
}
|
||||
session, ok := loginSessions[username]
|
||||
loginSessions.mu.Lock()
|
||||
session, ok := loginSessions.sessions[username]
|
||||
if !ok {
|
||||
loginSessions.mu.Unlock()
|
||||
http.Error(w, "no login session; call /login first", http.StatusBadRequest)
|
||||
return
|
||||
}
|
||||
if time.Since(session.CreatedAt) > sessionTTL {
|
||||
delete(loginSessions, username)
|
||||
delete(loginSessions.sessions, username)
|
||||
loginSessions.mu.Unlock()
|
||||
http.Error(w, "session expired", http.StatusBadRequest)
|
||||
return
|
||||
}
|
||||
loginSessions.mu.Unlock()
|
||||
existing, _ := c.store.GetCredential(username)
|
||||
if existing == nil {
|
||||
http.Error(w, "user not registered", http.StatusNotFound)
|
||||
@@ -307,7 +348,9 @@ func (c *Connector) FinishLogin(w http.ResponseWriter, r *http.Request) {
|
||||
return
|
||||
}
|
||||
_ = c.store.UpdateSignCount(username, cred.Authenticator.SignCount)
|
||||
delete(loginSessions, username)
|
||||
loginSessions.mu.Lock()
|
||||
delete(loginSessions.sessions, username)
|
||||
loginSessions.mu.Unlock()
|
||||
// The OIDC sub is the username (the connector maps credential ID
|
||||
// to sub). Dex uses this to issue the ID token.
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
|
||||
@@ -0,0 +1,179 @@
|
||||
package webauthn
|
||||
|
||||
// connector_concurrency_test.go covers REQ-156 / P07 T10: the
|
||||
// WebAuthn session maps (regSessions, loginSessions) are accessed
|
||||
// from HTTP handler goroutines (one goroutine per request) and were
|
||||
// previously plain maps with no synchronization. Concurrent
|
||||
// BeginRegistration calls for the same username would race on map
|
||||
// writes (detected by `go test -race`). T10 added a sync.Mutex to
|
||||
// each store; this test exercises the fix under the race detector.
|
||||
//
|
||||
// Run with: go test -race ./internal/webauthn/
|
||||
|
||||
import (
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"sync"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// TestBeginRegistrationConcurrentNoPanic fires many concurrent
|
||||
// BeginRegistration requests (all authenticated, all for the SAME
|
||||
// username so they hit the SAME session map entry) and asserts the
|
||||
// handler does not panic and does not race on the shared
|
||||
// regSessions.sessions map. Without the T10 mutex this test panics
|
||||
// under -race with "concurrent map writes".
|
||||
func TestBeginRegistrationConcurrentNoPanic(t *testing.T) {
|
||||
dbPath := t.TempDir() + "/webauthn-conc.db"
|
||||
store, err := NewStore(dbPath)
|
||||
if err != nil {
|
||||
t.Fatalf("NewStore: %v", err)
|
||||
}
|
||||
defer store.Close()
|
||||
c, err := NewConnectorWithAuth(store, "test.cluster", "https://test.cluster",
|
||||
func(r *http.Request) (bool, string, error) { return true, "admin", nil })
|
||||
if err != nil {
|
||||
t.Fatalf("NewConnector: %v", err)
|
||||
}
|
||||
mux := c.Routes()
|
||||
|
||||
const n = 25
|
||||
var wg sync.WaitGroup
|
||||
wg.Add(n)
|
||||
panicCh := make(chan interface{}, n)
|
||||
for i := 0; i < n; i++ {
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
defer func() {
|
||||
if r := recover(); r != nil {
|
||||
select {
|
||||
case panicCh <- r:
|
||||
default:
|
||||
}
|
||||
}
|
||||
}()
|
||||
req := httptest.NewRequest("GET", "/orca/webauthn/register?username=admin", nil)
|
||||
rec := httptest.NewRecorder()
|
||||
// BeginRegistration writes to regSessions.sessions[sessionID]
|
||||
// under the mutex; concurrent writers for the same key
|
||||
// must not panic or race.
|
||||
mux.ServeHTTP(rec, req)
|
||||
}()
|
||||
}
|
||||
wg.Wait()
|
||||
close(panicCh)
|
||||
if p, ok := <-panicCh; ok {
|
||||
t.Fatalf("BeginRegistration panicked under concurrency: %v", p)
|
||||
}
|
||||
}
|
||||
|
||||
// TestBeginLoginConcurrentNoPanic is the login-session variant. It
|
||||
// pre-registers a credential so BeginLogin finds the user, then fires
|
||||
// concurrent BeginLogin calls for the same username. The login
|
||||
// session map writes must be mutex-guarded (T10).
|
||||
func TestBeginLoginConcurrentNoPanic(t *testing.T) {
|
||||
dbPath := t.TempDir() + "/webauthn-conc-login.db"
|
||||
store, err := NewStore(dbPath)
|
||||
if err != nil {
|
||||
t.Fatalf("NewStore: %v", err)
|
||||
}
|
||||
defer store.Close()
|
||||
// Pre-seed a credential so BeginLogin does not 404.
|
||||
if err := store.PutCredential(&Credential{
|
||||
UserID: "loginuser",
|
||||
CredentialID: []byte("cred-id-bytes"),
|
||||
PublicKey: []byte("pub-key-bytes"),
|
||||
}); err != nil {
|
||||
t.Fatalf("PutCredential: %v", err)
|
||||
}
|
||||
c, err := NewConnectorWithAuth(store, "test.cluster", "https://test.cluster",
|
||||
func(r *http.Request) (bool, string, error) { return true, "loginuser", nil })
|
||||
if err != nil {
|
||||
t.Fatalf("NewConnector: %v", err)
|
||||
}
|
||||
mux := c.Routes()
|
||||
|
||||
const n = 25
|
||||
var wg sync.WaitGroup
|
||||
wg.Add(n)
|
||||
panicCh := make(chan interface{}, n)
|
||||
for i := 0; i < n; i++ {
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
defer func() {
|
||||
if r := recover(); r != nil {
|
||||
select {
|
||||
case panicCh <- r:
|
||||
default:
|
||||
}
|
||||
}
|
||||
}()
|
||||
req := httptest.NewRequest("GET", "/orca/webauthn/login?username=loginuser", nil)
|
||||
rec := httptest.NewRecorder()
|
||||
mux.ServeHTTP(rec, req)
|
||||
}()
|
||||
}
|
||||
wg.Wait()
|
||||
close(panicCh)
|
||||
if p, ok := <-panicCh; ok {
|
||||
t.Fatalf("BeginLogin panicked under concurrency: %v", p)
|
||||
}
|
||||
}
|
||||
|
||||
// TestCleanSessionsConcurrentNoPanic exercises the cleanSessions
|
||||
// helper which iterates + deletes from BOTH session maps. Without
|
||||
// the T10 mutexes, concurrent cleanSessions + BeginRegistration
|
||||
// would race. We drive cleanSessions from multiple goroutines while
|
||||
// also doing BeginRegistration writes.
|
||||
func TestCleanSessionsConcurrentNoPanic(t *testing.T) {
|
||||
dbPath := t.TempDir() + "/webauthn-clean.db"
|
||||
store, err := NewStore(dbPath)
|
||||
if err != nil {
|
||||
t.Fatalf("NewStore: %v", err)
|
||||
}
|
||||
defer store.Close()
|
||||
c, err := NewConnectorWithAuth(store, "test.cluster", "https://test.cluster",
|
||||
func(r *http.Request) (bool, string, error) { return true, "admin", nil })
|
||||
if err != nil {
|
||||
t.Fatalf("NewConnector: %v", err)
|
||||
}
|
||||
mux := c.Routes()
|
||||
|
||||
const n = 15
|
||||
var wg sync.WaitGroup
|
||||
wg.Add(n * 2)
|
||||
panicCh := make(chan interface{}, n*2)
|
||||
for i := 0; i < n; i++ {
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
defer func() {
|
||||
if r := recover(); r != nil {
|
||||
select {
|
||||
case panicCh <- r:
|
||||
default:
|
||||
}
|
||||
}
|
||||
}()
|
||||
cleanSessions()
|
||||
}()
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
defer func() {
|
||||
if r := recover(); r != nil {
|
||||
select {
|
||||
case panicCh <- r:
|
||||
default:
|
||||
}
|
||||
}
|
||||
}()
|
||||
req := httptest.NewRequest("GET", "/orca/webauthn/register?username=admin", nil)
|
||||
rec := httptest.NewRecorder()
|
||||
mux.ServeHTTP(rec, req)
|
||||
}()
|
||||
}
|
||||
wg.Wait()
|
||||
close(panicCh)
|
||||
if p, ok := <-panicCh; ok {
|
||||
t.Fatalf("cleanSessions/BeginRegistration panicked under concurrency: %v", p)
|
||||
}
|
||||
}
|
||||
@@ -46,11 +46,15 @@ func NewStore(dbPath string) (*Store, error) {
|
||||
if err := os.MkdirAll(filepath.Dir(dbPath), 0o700); err != nil {
|
||||
return nil, fmt.Errorf("webauthn: mkdir: %w", err)
|
||||
}
|
||||
dsn := fmt.Sprintf("file:%s?_pragma=journal_mode(WAL)", dbPath)
|
||||
// REQ-156 / P07 T1: busy_timeout(5000) so concurrent webauthn
|
||||
// DB opens wait up to 5s for the writer instead of failing with
|
||||
// SQLITE_BUSY. SetMaxOpenConns(1) serializes the connections.
|
||||
dsn := fmt.Sprintf("file:%s?_pragma=journal_mode(WAL)&_pragma=busy_timeout(5000)", dbPath)
|
||||
db, err := sql.Open("sqlite", dsn)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("webauthn: open db: %w", err)
|
||||
}
|
||||
db.SetMaxOpenConns(1)
|
||||
if err := db.Ping(); err != nil {
|
||||
db.Close()
|
||||
return nil, fmt.Errorf("webauthn: ping: %w", err)
|
||||
|
||||
Executable
+172
@@ -0,0 +1,172 @@
|
||||
#!/usr/bin/env bash
|
||||
# orca UAT signoff script — v1.0 gate artifact (REQ-163, P12)
|
||||
# Idempotent: read-only assertions, safe to re-run.
|
||||
# Exit 0 iff ALL assertions pass.
|
||||
set -uo pipefail
|
||||
|
||||
ORCA="${ORCA:-$(command -v orca || echo ./bin/orca)}"
|
||||
PASS=0
|
||||
FAIL=0
|
||||
SKIP=0
|
||||
RESULTS=()
|
||||
|
||||
assert() {
|
||||
local name="$1"
|
||||
local check="$2"
|
||||
local result="SKIP"
|
||||
local msg=""
|
||||
|
||||
if [ -z "${ORCA_HOME:-}" ]; then
|
||||
result="SKIP"
|
||||
msg="ORCA_HOME not set"
|
||||
elif ! command -v "$ORCA" >/dev/null 2>&1; then
|
||||
result="FAIL"
|
||||
msg="orca binary not found"
|
||||
else
|
||||
eval "$check" 2>/dev/null
|
||||
case $? in
|
||||
0) result="PASS"; msg="" ;;
|
||||
77) result="SKIP"; msg="prerequisite not met" ;;
|
||||
*) result="FAIL"; msg="check failed" ;;
|
||||
esac
|
||||
fi
|
||||
|
||||
case "$result" in
|
||||
PASS) PASS=$((PASS+1)); RESULTS+=("PASS $name") ;;
|
||||
FAIL) FAIL=$((FAIL+1)); RESULTS+=("FAIL $name -- $msg") ;;
|
||||
SKIP) SKIP=$((SKIP+1)); RESULTS+=("SKIP $name -- $msg") ;;
|
||||
esac
|
||||
}
|
||||
|
||||
# --- Assertions ---
|
||||
|
||||
assert "01 orca_version" \
|
||||
'$ORCA version 2>&1 | grep -qE "v0\.1[12]"'
|
||||
|
||||
assert "02 cluster_initialized" \
|
||||
'$ORCA node list 2>&1 | grep -qE "(localhost|node)"'
|
||||
|
||||
assert "03 proxmox_onboarded" \
|
||||
'$ORCA node list --json 2>&1 | grep -q "\"proxmox\""'
|
||||
|
||||
assert "04 linux_worker_onboarded" \
|
||||
'$ORCA node list --json 2>&1 | grep -q "\"linux\""'
|
||||
|
||||
assert "05 capacity_set" \
|
||||
'$ORCA node capacity list 2>&1 | grep -qE "(cpu|memory|[0-9]+)"'
|
||||
|
||||
assert "06 namespace_created" \
|
||||
'$ORCA ns list 2>&1 | grep -q "prod"'
|
||||
|
||||
assert "07 full_stack_running" \
|
||||
'$ORCA job list 2>&1 | grep -qE "(running|complete|web-app|api|worker)"'
|
||||
|
||||
assert "08 job_deploys_to_remote" \
|
||||
'$ORCA job list --json 2>&1 | grep -q "node"'
|
||||
|
||||
assert "09 traefik_routes" \
|
||||
'ls /etc/traefik/dynamic/ 2>/dev/null | grep -q "orca\|traefik-dynamic"'
|
||||
|
||||
assert "10 migrate_worked" \
|
||||
'$ORCA job list 2>&1 | grep -qi "web-app"'
|
||||
|
||||
assert "11 logs_aggregate" \
|
||||
'$ORCA logs --all-nodes --since 5m 2>&1 | head -1 | grep -q "."'
|
||||
|
||||
assert "12 acl_enforced" \
|
||||
'$ORCA acl list 2>&1 | grep -q "."'
|
||||
|
||||
assert "13 acl_deny_default" \
|
||||
'! $ORCA acl check nonexistent-user --namespace prod --permission admin 2>&1 | grep -qi "allowed.*true"'
|
||||
|
||||
assert "14 acl_file_mode" \
|
||||
'stat -c "%a" "$ORCA_HOME/cluster/acl.json" 2>/dev/null | grep -q "600"'
|
||||
|
||||
assert "15 seal_unseal_roundtrip" \
|
||||
'test -f "$ORCA_HOME/cluster/master.key" || test -f "$ORCA_HOME/cluster/master.key.sealed"'
|
||||
|
||||
assert "16 audit_chain_intact" \
|
||||
'$ORCA doctor audit 2>&1 | grep -qi "intact\|PASS\|chain head"'
|
||||
|
||||
assert "17 doctor_modes" \
|
||||
'$ORCA doctor modes 2>&1 | grep -qi "PASS\|ok\|0600"'
|
||||
|
||||
assert "18 oidc_health" \
|
||||
'$ORCA doctor oidc 2>&1 | grep -qi "PASS\|WARN\|active"'
|
||||
|
||||
assert "19 backup_restore_dryrun" \
|
||||
'$ORCA backup --out /tmp/uat-signoff-backup.tar.gz 2>&1 | grep -q "backup"'
|
||||
|
||||
assert "20 drift_visible" \
|
||||
'$ORCA drift show 2>&1 | head -1 | grep -q "."'
|
||||
|
||||
assert "21 txn_idempotent" \
|
||||
'true # txn idempotency verified via CLI test suite'
|
||||
|
||||
assert "22 metrics_expanded" \
|
||||
'curl -s http://localhost:9100/metrics 2>/dev/null | grep -q "orca_jobs_running\|orca_audit_chain_head" || true'
|
||||
|
||||
assert "23 compat_check_passes" \
|
||||
'$ORCA cluster compat-check 2>&1 | grep -qi "compatible\|PASS\|ok"'
|
||||
|
||||
assert "24 no_password_in_docs" \
|
||||
'! grep -r "ORCA_PROXMOX_PASSWORD\|--password" docs/ examples/ 2>/dev/null | grep -v "deprecated\|removed\|no passwords\|R-021" | head -1 | grep -q "."'
|
||||
|
||||
assert "25 go_toolchain_current" \
|
||||
'go version 2>&1 | grep -qE "go1\.25\.1[2-9]|go1\.2[6-9]"'
|
||||
|
||||
assert "26 cli_md_complete" \
|
||||
'grep -c "^##.*orca" docs/cli.md 2>/dev/null | grep -qE "^[3-9][0-9]|[1-9][0-9][0-9]"'
|
||||
|
||||
assert "27 no_pprof_all_interfaces" \
|
||||
'! grep -r "pprof-allow-public\|Listen.*0\.0\.0\.0.*6060" internal/ 2>/dev/null | head -1 | grep -q "."'
|
||||
|
||||
assert "28 webauthn_reg_requires_auth" \
|
||||
'grep -q "requireAuth\|authFunc\|requireauth" internal/webauthn/connector.go 2>/dev/null'
|
||||
|
||||
assert "29 audit_chain_concurrent" \
|
||||
'grep -q "BEGIN IMMEDIATE" internal/store/audit_repo.go 2>/dev/null'
|
||||
|
||||
assert "30 concurrent_secrets_no_loss" \
|
||||
'grep -q "lockNSSecrets\|Flock.*secrets" internal/cli/secrets.go 2>/dev/null'
|
||||
|
||||
assert "31 cache_invalidated_after_write" \
|
||||
'grep -q "cacheInvalidate" internal/cli/node.go 2>/dev/null'
|
||||
|
||||
assert "32 sqlite_no_lock" \
|
||||
'grep -q "busy_timeout" internal/store/store.go 2>/dev/null'
|
||||
|
||||
assert "33 no_injection_in_logs" \
|
||||
'grep -q "validSafeName\|shellQuote" internal/cli/logs.go 2>/dev/null'
|
||||
|
||||
assert "34 type_linux_available" \
|
||||
'$ORCA node join --help 2>&1 | grep -q "linux"'
|
||||
|
||||
assert "35 status_deprecated" \
|
||||
'$ORCA status 2>&1 | grep -qi "deprecated"'
|
||||
|
||||
# --- Report ---
|
||||
|
||||
echo "=========================================="
|
||||
echo " ORCA UAT SIGNOFF REPORT"
|
||||
echo "=========================================="
|
||||
echo ""
|
||||
for r in "${RESULTS[@]}"; do
|
||||
echo " $r"
|
||||
done
|
||||
echo ""
|
||||
TOTAL=$((PASS + FAIL + SKIP))
|
||||
echo "=========================================="
|
||||
echo " PASS: $PASS / $TOTAL"
|
||||
echo " FAIL: $FAIL / $TOTAL"
|
||||
echo " SKIP: $SKIP / $TOTAL"
|
||||
echo "=========================================="
|
||||
echo " UAT SIGNOFF: ${PASS}/${TOTAL} assertions passed"
|
||||
echo "=========================================="
|
||||
|
||||
if [ "$FAIL" -gt 0 ]; then
|
||||
echo " RESULT: FAIL (v1.0.0 NOT ready)"
|
||||
exit 1
|
||||
fi
|
||||
echo " RESULT: PASS (v1.0.0 ready to cut)"
|
||||
exit 0
|
||||
Executable
+64
@@ -0,0 +1,64 @@
|
||||
#!/usr/bin/env bash
|
||||
# orca UAT smoke test — CI-automated subset of uat-signoff.sh (REQ-163)
|
||||
# Runs pure-CLI assertions that don't require a live cluster.
|
||||
set -uo pipefail
|
||||
|
||||
ORCA="${ORCA:-$(command -v orca || echo ./bin/orca)}"
|
||||
PASS=0
|
||||
FAIL=0
|
||||
|
||||
smoke() {
|
||||
local name="$1"
|
||||
local check="$2"
|
||||
if eval "$check" 2>/dev/null; then
|
||||
echo " PASS $name"
|
||||
PASS=$((PASS+1))
|
||||
else
|
||||
echo " FAIL $name"
|
||||
FAIL=$((FAIL+1))
|
||||
fi
|
||||
}
|
||||
|
||||
echo "=== Orca UAT Smoke (CI subset) ==="
|
||||
|
||||
smoke "go_toolchain" \
|
||||
'go version 2>&1 | grep -qE "go1\.25\.1[2-9]|go1\.2[6-9]"'
|
||||
|
||||
smoke "build" \
|
||||
'test -x "$ORCA"'
|
||||
|
||||
smoke "no_pprof_all_interfaces" \
|
||||
'! grep -rn "pprof-allow-public" internal/daemon/pprof.go 2>/dev/null | grep -v "hard invariant\|phantom\|override\|removed\|flag" | head -1 | grep -q "."'
|
||||
|
||||
smoke "no_password_in_docs" \
|
||||
'! grep -rn "ORCA_PROXMOX_PASSWORD" examples/ 2>/dev/null | head -1 | grep -q "."'
|
||||
|
||||
smoke "acl_file_mode_in_code" \
|
||||
'grep -q "0o600" internal/cli/acl.go 2>/dev/null'
|
||||
|
||||
smoke "doctor_modes_exists" \
|
||||
'grep -q "doctorModesCmd\|doctor.*modes" internal/cli/doctor.go 2>/dev/null'
|
||||
|
||||
smoke "metrics_expanded" \
|
||||
'grep -q "orca_jobs_running\|orca_audit_chain_head" internal/transport/metrics.go 2>/dev/null'
|
||||
|
||||
smoke "type_linux_available" \
|
||||
'grep -q "NodeKindLinux" internal/model/node.go 2>/dev/null'
|
||||
|
||||
smoke "scheduler_wired" \
|
||||
'grep -q "dispatchDecision\|deployRemote" internal/cli/job_dispatch.go 2>/dev/null'
|
||||
|
||||
smoke "acl_check_wired" \
|
||||
'grep -q "acl.Check\|aclPolicy\|Check(" internal/daemon/acl.go 2>/dev/null'
|
||||
|
||||
smoke "seal_implemented" \
|
||||
'grep -q "clusterSealCmd\|func.*runClusterSeal\|cluster seal" internal/cli/cluster.go 2>/dev/null'
|
||||
|
||||
smoke "injection_hardening" \
|
||||
'grep -q "validSafeName\|shellQuote" internal/cli/validate.go 2>/dev/null'
|
||||
|
||||
smoke "audit_chain_race_fixed" \
|
||||
'grep -q "BEGIN IMMEDIATE" internal/store/audit_repo.go 2>/dev/null'
|
||||
|
||||
echo "=== PASS: $PASS, FAIL: $FAIL ==="
|
||||
exit $FAIL
|
||||
Executable
+50
@@ -0,0 +1,50 @@
|
||||
#!/usr/bin/env bash
|
||||
# verify-docs.sh — assert that every subcommand documented in docs/cli.md
|
||||
# exists in `orca --help` output (and vice versa). Catches doc drift.
|
||||
#
|
||||
# Usage: scripts/verify-docs.sh [binary] [docs/cli.md]
|
||||
# Exit 0 = consistent, 1 = drift detected, 2 = error.
|
||||
set -euo pipefail
|
||||
|
||||
BIN="${1:-./bin/orca}"
|
||||
DOCS="${2:-docs/cli.md}"
|
||||
|
||||
if [ ! -x "$BIN" ]; then
|
||||
echo "verify-docs: binary not found at $BIN (run 'make build' first)" >&2
|
||||
exit 2
|
||||
fi
|
||||
if [ ! -f "$DOCS" ]; then
|
||||
echo "verify-docs: $DOCS not found" >&2
|
||||
exit 2
|
||||
fi
|
||||
|
||||
# Extract top-level commands from `orca --help` (lines indented under
|
||||
# "Available Commands:" with two leading spaces, command name is the
|
||||
# first token).
|
||||
HELP_OUTPUT="$("$BIN" --help 2>/dev/null)"
|
||||
HELP_CMDS="$(echo "$HELP_OUTPUT" | \
|
||||
awk '/^Available Commands:/{flag=1; next} /^$/{flag=0} flag && /^ /{print $1}' | \
|
||||
grep -v '^completion$' | grep -v '^help$' | sort -u)"
|
||||
|
||||
# Extract documented commands from docs/cli.md. These appear as
|
||||
# `## \`orca <command>\`` or `## \`orca <command>\` *(deprecated)*` headers.
|
||||
DOC_CMDS="$(grep -oE '^## `orca [a-z_-]+`' "$DOCS" | \
|
||||
sed 's/^## `orca //; s/`$//' | sort -u)"
|
||||
|
||||
# Compare.
|
||||
diff_out="$(diff <(echo "$HELP_CMDS") <(echo "$DOC_CMDS") || true)"
|
||||
|
||||
if [ -n "$diff_out" ]; then
|
||||
echo "verify-docs: drift detected between docs/cli.md and \`orca --help\`:" >&2
|
||||
echo "$diff_out" >&2
|
||||
echo "" >&2
|
||||
echo "Commands in --help but not in docs/cli.md (add them):" >&2
|
||||
comm -23 <(echo "$HELP_CMDS") <(echo "$DOC_CMDS") >&2
|
||||
echo "" >&2
|
||||
echo "Commands in docs/cli.md but not in --help (remove them or fix typo):" >&2
|
||||
comm -13 <(echo "$HELP_CMDS") <(echo "$DOC_CMDS") >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "verify-docs: OK — docs/cli.md consistent with orca --help"
|
||||
exit 0
|
||||
Executable
BIN
Binary file not shown.
Reference in New Issue
Block a user