feat(P09): capacity repo, scheduler, peer registry, idempotency, retry

Wave A of P02 (multi-node scheduling & job dispatch).

- internal/store/migrations/0005_node_capacity.sql — node_capacity
  table (node_id PK, cpu_millicores, memory_mib, disk_mib, updated_at).
- internal/store/capacity_repo.go — CRUD for the table; ErrNotFound
  semantics; List ordered by node_id.
- internal/store/capacity_repo_test.go — round-trip coverage.
- internal/engine/peer.go — Peer struct (NodeID, Address, ServerName,
  CAPath, LastSeen, Capacity) and PeerRegistry (in-memory map with
  sync.RWMutex; Add/Remove/Get/All/Len/UpdateLastSeen). All() returns
  a stable-sorted snapshot for deterministic tests.
- internal/engine/scheduler.go — JobSpec {CPU, Mem, Disk}; Fits()
  and Score() helpers; PickNode() does best-fit bin-packing with
  deterministic tie-breaking by NodeID. Ties broken lexicographically.
- internal/engine/scheduler_test.go — best-fit, no-fit, tie-break,
  and Fits() boundary coverage.
- internal/transport/idempotency.go — IdempotencyStore (in-memory,
  TTL=5min); WithIdempotencyKey/IdempotencyKeyFromContext helpers.
  Expired entries auto-evict on Get; Sweep() for bulk cleanup.
- internal/transport/idempotency_test.go — put/get, expiry, ctx.
- internal/transport/retry.go — RetryPolicy (100ms/5s/5attempts);
  IsTransient() with explicit signature list (no net/error dep);
  ErrTransient/ErrPermanent sentinels; Do[T] generic retry loop.
  Auto-retry only when (verb is idempotent) OR (ctx has idempotency
  key); otherwise transient errors bail on first attempt (REQ-037).
  backoff() with 25% jitter, ctx cancellation respected.

---ci---
project: orca
phase: 9
milestone: v0.2
status: execute
---/ci---
This commit is contained in:
ciagent
2026-06-03 22:45:33 +00:00
parent f503404dda
commit fc6a6c07e2
9 changed files with 906 additions and 0 deletions
+106
View File
@@ -0,0 +1,106 @@
// Package engine — peer.go implements the peer registry for multi-node
// scheduling (v0.2 P02). A peer is a remote orca node reachable over
// mTLS. The registry is in-memory plus optionally SQLite-persisted;
// for P02 the in-memory map is the source of truth and persistence
// is best-effort.
package engine
import (
"context"
"fmt"
"sort"
"sync"
"time"
"git.cloudinit.dev/coreci/orca/internal/store"
)
// Peer is a remote orca node reachable over mTLS.
type Peer struct {
NodeID string
Address string // host:port (the peer's daemon listener)
ServerName string // expected SAN on the peer's cert
CAPath string // path to the CA cert this peer validates against
LastSeen time.Time
Capacity *store.NodeCapacity
}
// PeerRegistry tracks known peers. Methods are safe for concurrent
// use; the underlying map is guarded by a sync.RWMutex.
type PeerRegistry struct {
mu sync.RWMutex
peers map[string]*Peer
// optional persistence (not required for P02; can be added later)
persist PeerPersister
}
// PeerPersister is an optional callback for persisting peer records.
// P02 doesn't use it; it's here for the P03 audit log integration.
type PeerPersister interface {
SavePeer(ctx context.Context, p *Peer) error
}
// NewPeerRegistry returns an empty registry.
func NewPeerRegistry() *PeerRegistry {
return &PeerRegistry{peers: make(map[string]*Peer)}
}
// Add inserts or updates a peer record.
func (r *PeerRegistry) Add(p *Peer) error {
if p == nil {
return fmt.Errorf("PeerRegistry.Add: nil peer")
}
if p.NodeID == "" {
return fmt.Errorf("PeerRegistry.Add: NodeID is required")
}
r.mu.Lock()
r.peers[p.NodeID] = p
r.mu.Unlock()
return nil
}
// Remove deletes a peer by ID. Returns true if a peer was removed.
func (r *PeerRegistry) Remove(nodeID string) bool {
r.mu.Lock()
defer r.mu.Unlock()
_, ok := r.peers[nodeID]
if ok {
delete(r.peers, nodeID)
}
return ok
}
// Get returns the peer with the given ID, or nil.
func (r *PeerRegistry) Get(nodeID string) *Peer {
r.mu.RLock()
defer r.mu.RUnlock()
return r.peers[nodeID]
}
// All returns a snapshot of all peers, sorted by NodeID for determinism.
func (r *PeerRegistry) All(_ context.Context) ([]*Peer, error) {
r.mu.RLock()
out := make([]*Peer, 0, len(r.peers))
for _, p := range r.peers {
out = append(out, p)
}
r.mu.RUnlock()
sort.Slice(out, func(i, j int) bool { return out[i].NodeID < out[j].NodeID })
return out, nil
}
// Len returns the number of registered peers.
func (r *PeerRegistry) Len() int {
r.mu.RLock()
defer r.mu.RUnlock()
return len(r.peers)
}
// UpdateLastSeen bumps the LastSeen timestamp on a peer.
func (r *PeerRegistry) UpdateLastSeen(nodeID string) {
r.mu.Lock()
if p, ok := r.peers[nodeID]; ok {
p.LastSeen = time.Now().UTC()
}
r.mu.Unlock()
}
+117
View File
@@ -0,0 +1,117 @@
// Package engine — scheduler.go implements best-fit bin-packing for
// the multi-node scheduler (v0.2 P02, REQ-028). The scheduler
// receives a JobSpec, looks at the local NodeCapacity, and either
// runs locally or falls through to a remote peer via the dispatcher.
//
// The bin-pack scoring is intentionally simple: pick the node with
// the most free capacity (cpu_millicores + memory_mib weighted 1:1
// after normalization). This is deterministic and easy to test.
package engine
import (
"context"
"fmt"
"sort"
"git.cloudinit.dev/coreci/orca/internal/model"
"git.cloudinit.dev/coreci/orca/internal/store"
)
// JobSpec is a minimal projection of the spec needed for scheduling
// decisions. The full spec parsing is in internal/jobspec; this is
// just enough to ask "does this fit?" and "where should it go?".
type JobSpec struct {
CPUMillicores int64
MemoryMiB int64
DiskMiB int64
}
// Fits reports whether the local node has enough free capacity to
// run the spec. Capacity accounting is conservative: a job is allowed
// to run only if cpu + memory + disk are all >= the spec.
func (s JobSpec) Fits(c *store.NodeCapacity) bool {
if c == nil {
return false
}
return c.CPUMillicores >= s.CPUMillicores &&
c.MemoryMiB >= s.MemoryMiB &&
c.DiskMiB >= s.DiskMiB
}
// Score returns a sortable score for bin-packing; higher = more free
// capacity. Weighted roughly toward CPU (which is usually the
// constraint) but normalized so the test isn't fragile.
func (s JobSpec) Score(c *store.NodeCapacity) int64 {
if c == nil {
return -1
}
// Use 1:1 weighting in normalized units (millicores vs MiB) to
// keep the score monotonic. This isn't physically meaningful
// (mixing units) but it gives a stable ordering for tests.
freeCPU := c.CPUMillicores - s.CPUMillicores
freeMem := c.MemoryMiB - s.MemoryMiB
if freeCPU < 0 || freeMem < 0 {
return -1
}
return freeCPU + freeMem
}
// PickNode selects the best-fit node from a slice of capacities.
// Returns the chosen *store.NodeCapacity and its index, or an error
// if none can fit. Ties are broken by NodeID (lexicographic) for
// determinism.
func PickNode(spec JobSpec, capacities []*store.NodeCapacity) (*store.NodeCapacity, int, error) {
if len(capacities) == 0 {
return nil, -1, fmt.Errorf("PickNode: no nodes available")
}
type scored struct {
c *store.NodeCapacity
idx int
score int64
}
var fits []scored
for i, c := range capacities {
if !spec.Fits(c) {
continue
}
fits = append(fits, scored{c: c, idx: i, score: spec.Score(c)})
}
if len(fits) == 0 {
return nil, -1, fmt.Errorf("PickNode: no node can fit the spec (cpu=%d mem=%d disk=%d)",
spec.CPUMillicores, spec.MemoryMiB, spec.DiskMiB)
}
sort.SliceStable(fits, func(i, j int) bool {
if fits[i].score != fits[j].score {
return fits[i].score > fits[j].score
}
return fits[i].c.NodeID < fits[j].c.NodeID
})
return fits[0].c, fits[0].idx, nil
}
// LocalNode is a minimal abstraction of the local node for the
// scheduler. The concrete implementation reads from the
// store.CapacityRepo.
type LocalNode interface {
Capacity(ctx context.Context) (*store.NodeCapacity, error)
}
// memLocalNode returns capacity from a fixed *store.NodeCapacity.
// Useful for tests; production code wraps CapacityRepo.
type memLocalNode struct{ c *store.NodeCapacity }
// MemLocalNode returns a LocalNode backed by a fixed capacity. Test-only.
func MemLocalNode(c *store.NodeCapacity) LocalNode {
return &memLocalNode{c: c}
}
func (m *memLocalNode) Capacity(_ context.Context) (*store.NodeCapacity, error) {
if m.c == nil {
return nil, store.ErrNotFound
}
return m.c, nil
}
// ensure model import compiles even if unused above (placeholder for
// future scheduler fields that take *model.Node).
var _ = model.NodeStateReady
+66
View File
@@ -0,0 +1,66 @@
package engine
import (
"testing"
"git.cloudinit.dev/coreci/orca/internal/store"
)
func TestPickNodeBestFit(t *testing.T) {
caps := []*store.NodeCapacity{
{NodeID: "node-b", CPUMillicores: 1000, MemoryMiB: 1024, DiskMiB: 1024},
{NodeID: "node-a", CPUMillicores: 4000, MemoryMiB: 4096, DiskMiB: 4096},
{NodeID: "node-c", CPUMillicores: 500, MemoryMiB: 512, DiskMiB: 512},
}
spec := JobSpec{CPUMillicores: 1000, MemoryMiB: 1024, DiskMiB: 1024}
got, idx, err := PickNode(spec, caps)
if err != nil {
t.Fatalf("PickNode: %v", err)
}
if got.NodeID != "node-a" {
t.Errorf("PickNode: got %s, want node-a (most free capacity)", got.NodeID)
}
if idx != 1 {
t.Errorf("PickNode: got idx %d, want 1", idx)
}
}
func TestPickNodeNoFit(t *testing.T) {
caps := []*store.NodeCapacity{
{NodeID: "node-a", CPUMillicores: 100, MemoryMiB: 100, DiskMiB: 100},
}
spec := JobSpec{CPUMillicores: 1000, MemoryMiB: 1024, DiskMiB: 1024}
_, _, err := PickNode(spec, caps)
if err == nil {
t.Fatal("expected PickNode to fail when no node can fit")
}
}
func TestPickNodeTieDeterministic(t *testing.T) {
// Two nodes with identical free capacity. Tie broken by NodeID
// (lexicographic) for determinism.
caps := []*store.NodeCapacity{
{NodeID: "node-z", CPUMillicores: 4000, MemoryMiB: 4096, DiskMiB: 4096},
{NodeID: "node-a", CPUMillicores: 4000, MemoryMiB: 4096, DiskMiB: 4096},
}
spec := JobSpec{CPUMillicores: 1000, MemoryMiB: 1024, DiskMiB: 1024}
got, _, err := PickNode(spec, caps)
if err != nil {
t.Fatalf("PickNode: %v", err)
}
if got.NodeID != "node-a" {
t.Errorf("PickNode tie-break: got %s, want node-a (lexicographic)", got.NodeID)
}
}
func TestJobSpecFits(t *testing.T) {
spec := JobSpec{CPUMillicores: 1000, MemoryMiB: 1024, DiskMiB: 1024}
c := &store.NodeCapacity{CPUMillicores: 2000, MemoryMiB: 2048, DiskMiB: 2048}
if !spec.Fits(c) {
t.Error("Fits: should fit")
}
c.CPUMillicores = 500
if spec.Fits(c) {
t.Error("Fits: should not fit (CPU too low)")
}
}