Compare commits
6 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 3a76a32964 | |||
| 872ffcaf25 | |||
| 2c53ad6213 | |||
| c3819dde12 | |||
| fb85898569 | |||
| c10779873b |
@@ -1 +1 @@
|
||||
{ "phase": "P03/P04/P08", "stage": "verify", "milestone": "v0.9", "phase_role": "execution", "updated_at": "2026-08-05T04:05:00Z", "milestone_complete": false, "verify": { "build": "pass", "go_test": "22/22", "bats": "20/20", "gofmt": "clean", "verify_reqs": "90 consistent" } }
|
||||
{ "phase": "P07a/b/c", "stage": "verify", "milestone": "v0.9", "phase_role": "execution", "updated_at": "2026-08-05T04:40:00Z", "milestone_complete": false, "gates_cleared_this_phase": ["C-01"], "verify": { "build": "pass", "go_test": "23/23", "bats": "20/20", "gofmt": "clean", "verify_reqs": "90 consistent" } }
|
||||
|
||||
@@ -445,3 +445,4 @@ are recorded in `REQUIREMENTS.md`. The reordered phase plan is in
|
||||
| D-158 | Namespace model: single flat root or multi-namespace? | **Multi-namespace under ORCA_HOME (R-002)** | Hard multi-tenant product requirement (override ground 3). `_defaults/` implicit root; `cluster/` for cluster-wide; per-namespace `db/`, `.env`, `.env.secrets`, `jobs/`, `alloc/`, `ns.md`. No namespace column in SQLite. | 0.84 |
|
||||
| D-179 | Jobspec format: HCL canonical (AD-007) or Markdown? | **Markdown with YAML frontmatter canonical (R-013); HCL legacy** | PRD §8 — Markdown + body preservation is the operator-facing format. HCL adapter (REQ-064) preserves `orca job run old-spec.hcl` during migration. | 0.85 |
|
||||
| D-185 | Re-architecture justification: incremental additive or full re-architecture? | **Full re-architecture (overridden by user)** | Six-part evidence basis above; the grill's REPLAN mechanics (PC-01..PC-10, C-01..C-19) adopted as gates. The incremental-additive path was evaluated and rejected on grounds 1 + 5 (daemon failing; SSH-push only viable). | 0.88 |
|
||||
| D-187 | wasmtime Go binding (bytecodealliance/wasmtime-go) is CGO-based — does adopting it revoke D-002 (modernc/sqlite CGO-free cross-compile story)? | **Use the wasmtime CLI (apt-installed on peer) via SSH exec; do NOT import wasmtime-go.** | The Go binding links libwasmtime via cgo and would revoke D-002's CGO-free cross-compile story. The CLI-via-SSH approach (same pattern as podman/qm/pct) avoids CGO entirely. `internal/runtime/wasm.go` imports only stdlib + sshpush. `CGO_ENABLED=0 go build ./...` succeeds. C-01 grill gate SATISFIED; D-002 NOT revoked. Full evaluation in `internal/runtime/C01_WASMTIME_CGO_EVAL.md`. | 0.90 |
|
||||
|
||||
+118
-7
@@ -42,13 +42,26 @@ type SystemdEmitter struct{}
|
||||
const unitNamePrefix = "orca-v1-"
|
||||
|
||||
// Render renders the systemd unit file for a process-runtime workload.
|
||||
// The unit name is /etc/systemd/system/<unitNamePrefix><spec.Name>.service
|
||||
// and the content is a [Service] block with ExecStart, optional
|
||||
// ExecStartPost (lifecycle.post_start), optional ExecStop
|
||||
// (lifecycle.pre_stop), and the R-007 socket-plumbing lines
|
||||
// (RuntimeDirectory=, optional TCP-bind ExecStartPre). Mode is 0644.
|
||||
//
|
||||
// The rendered shape is:
|
||||
// When the spec has no Tasks (the single-process case, the historical
|
||||
// shape), the unit name is
|
||||
// /etc/systemd/system/<unitNamePrefix><spec.Name>.service and the
|
||||
// content is a [Service] block with ExecStart, optional ExecStartPost
|
||||
// (lifecycle.post_start), optional ExecStop (lifecycle.pre_stop), and
|
||||
// the R-007 socket-plumbing lines (RuntimeDirectory=, optional
|
||||
// TCP-bind ExecStartPre). Mode is 0644.
|
||||
//
|
||||
// When the spec has a task group (P06, spec.Tasks non-empty), the
|
||||
// alloc is multi-process and Render emits one systemd unit per task
|
||||
// (`orca-v1-alloc-<alloc-id>-<task-name>.service`) plus a single
|
||||
// grouping target unit (`orca-v1-alloc-<alloc-id>.target`) that
|
||||
// starts/stops all tasks together. Each per-task unit carries
|
||||
// `PartOf=orca-v1-alloc-<alloc-id>.target` and is
|
||||
// `WantedBy=multi-user.target` so the task starts at boot. Tasks
|
||||
// that omit their own runtime inherit the top-level spec.Runtime as
|
||||
// the per-group default.
|
||||
//
|
||||
// The rendered shape (single-process) is:
|
||||
//
|
||||
// [Service]
|
||||
// ExecStart=<runtime command>
|
||||
@@ -62,7 +75,8 @@ const unitNamePrefix = "orca-v1-"
|
||||
//
|
||||
// Returns an error if the spec is nil, the spec is missing its name,
|
||||
// the runtime block is nil, or the runtime command is empty (a
|
||||
// workload with no command has nothing to ExecStart).
|
||||
// workload with no command has nothing to ExecStart). For task groups,
|
||||
// returns an error if any task has no resolvable runtime command.
|
||||
func (SystemdEmitter) Render(spec *jobspec.WorkloadSpec, node *Node) ([]File, error) {
|
||||
if spec == nil {
|
||||
return nil, errors.New("emitter/systemd: spec is nil")
|
||||
@@ -70,6 +84,9 @@ func (SystemdEmitter) Render(spec *jobspec.WorkloadSpec, node *Node) ([]File, er
|
||||
if strings.TrimSpace(spec.Name) == "" {
|
||||
return nil, errors.New("emitter/systemd: spec name is empty")
|
||||
}
|
||||
if len(spec.Tasks) > 0 {
|
||||
return renderTaskGroup(spec, node)
|
||||
}
|
||||
if spec.Runtime == nil {
|
||||
return nil, errors.New("emitter/systemd: runtime block is nil")
|
||||
}
|
||||
@@ -81,6 +98,100 @@ func (SystemdEmitter) Render(spec *jobspec.WorkloadSpec, node *Node) ([]File, er
|
||||
return []File{{Path: path, Content: content, Mode: "0644"}}, nil
|
||||
}
|
||||
|
||||
// renderTaskGroup renders one systemd unit per task plus the grouping
|
||||
// target unit. Each task's runtime falls back to the top-level
|
||||
// spec.Runtime when the task omits its own. Tasks with no resolvable
|
||||
// command (no task.Command, no task.Runtime.Command, no top-level
|
||||
// Runtime) return an error.
|
||||
func renderTaskGroup(spec *jobspec.WorkloadSpec, node *Node) ([]File, error) {
|
||||
allocID := spec.Name
|
||||
targetUnit := fmt.Sprintf("%salloc-%s.target", unitNamePrefix, allocID)
|
||||
targetPath := fmt.Sprintf("/etc/systemd/system/%s", targetUnit)
|
||||
var files []File
|
||||
for _, task := range spec.Tasks {
|
||||
rt := taskRuntime(spec, &task)
|
||||
if rt == nil {
|
||||
return nil, fmt.Errorf("emitter/systemd: task %q has no runtime (set tasks[].runtime or top-level runtime)", task.Name)
|
||||
}
|
||||
cmd := taskCommand(spec, &task, rt)
|
||||
if strings.TrimSpace(cmd) == "" {
|
||||
return nil, fmt.Errorf("emitter/systemd: task %q command is empty", task.Name)
|
||||
}
|
||||
unitName := fmt.Sprintf("%salloc-%s-%s.service", unitNamePrefix, allocID, task.Name)
|
||||
path := fmt.Sprintf("/etc/systemd/system/%s", unitName)
|
||||
content := renderTaskUnit(spec, &task, rt, cmd, targetUnit)
|
||||
files = append(files, File{Path: path, Content: content, Mode: "0644"})
|
||||
}
|
||||
files = append(files, File{
|
||||
Path: targetPath,
|
||||
Content: renderTargetUnit(targetUnit, spec, allocID),
|
||||
Mode: "0644",
|
||||
})
|
||||
return files, nil
|
||||
}
|
||||
|
||||
// taskRuntime returns the effective runtime for a task: the task's own
|
||||
// runtime when set, otherwise the top-level spec.Runtime (the per-group
|
||||
// default). Returns nil when neither is set.
|
||||
func taskRuntime(spec *jobspec.WorkloadSpec, task *jobspec.TaskGroupTask) *jobspec.RuntimeBlock {
|
||||
if task.Runtime != nil {
|
||||
return task.Runtime
|
||||
}
|
||||
return spec.Runtime
|
||||
}
|
||||
|
||||
// taskCommand returns the ExecStart command for a task. A task-level
|
||||
// Command takes precedence; otherwise the task's runtime command is
|
||||
// used; otherwise the top-level runtime command is used. Returns an
|
||||
// empty string when none is set.
|
||||
func taskCommand(spec *jobspec.WorkloadSpec, task *jobspec.TaskGroupTask, rt *jobspec.RuntimeBlock) string {
|
||||
if strings.TrimSpace(task.Command) != "" {
|
||||
return task.Command
|
||||
}
|
||||
if rt != nil && strings.TrimSpace(rt.Command) != "" {
|
||||
return rt.Command
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// renderTaskUnit renders a single per-task systemd [Unit]+[Service]
|
||||
// block. The unit is `PartOf=` the alloc target and
|
||||
// `WantedBy=multi-user.target` so it starts at boot and stops with the
|
||||
// group. The [Service] block carries the task's ExecStart and the
|
||||
// socket-plumbing lines derived from the spec's ports.
|
||||
func renderTaskUnit(spec *jobspec.WorkloadSpec, task *jobspec.TaskGroupTask, rt *jobspec.RuntimeBlock, cmd, targetUnit string) string {
|
||||
var b strings.Builder
|
||||
b.WriteString("[Unit]\n")
|
||||
b.WriteString(fmt.Sprintf("Description=orca alloc task %s\n", task.Name))
|
||||
b.WriteString(fmt.Sprintf("PartOf=%s\n", targetUnit))
|
||||
b.WriteString("\n[Service]\n")
|
||||
b.WriteString(fmt.Sprintf("ExecStart=%s\n", cmd))
|
||||
for _, line := range (SocketEmitter{}).RenderSocketLines(spec) {
|
||||
b.WriteString(line)
|
||||
b.WriteString("\n")
|
||||
}
|
||||
b.WriteString("\n[Install]\n")
|
||||
b.WriteString("WantedBy=multi-user.target\n")
|
||||
return b.String()
|
||||
}
|
||||
|
||||
// renderTargetUnit renders the grouping target unit
|
||||
// (`orca-v1-alloc-<alloc-id>.target`) that starts/stops all tasks
|
||||
// together. The [Unit] block lists every per-task unit under Wants=
|
||||
// so `systemctl start <target>` brings them all up, and
|
||||
// `systemctl stop <target>` tears them down (PartOf= propagates stop).
|
||||
func renderTargetUnit(targetUnit string, spec *jobspec.WorkloadSpec, allocID string) string {
|
||||
var b strings.Builder
|
||||
b.WriteString("[Unit]\n")
|
||||
b.WriteString(fmt.Sprintf("Description=orca alloc %s task group\n", allocID))
|
||||
for _, task := range spec.Tasks {
|
||||
b.WriteString(fmt.Sprintf("Wants=%salloc-%s-%s.service\n", unitNamePrefix, allocID, task.Name))
|
||||
}
|
||||
b.WriteString("\n[Install]\n")
|
||||
b.WriteString("WantedBy=multi-user.target\n")
|
||||
return b.String()
|
||||
}
|
||||
|
||||
// renderSystemdUnit renders the full [Service] block for the spec,
|
||||
// including ExecStart, lifecycle hooks (ExecStartPost, ExecStop), and
|
||||
// the R-007 socket-plumbing lines (RuntimeDirectory=, optional
|
||||
|
||||
@@ -154,3 +154,190 @@ func TestSystemdEmitter_UnitNamePrefix(t *testing.T) {
|
||||
t.Errorf("unitNamePrefix = %q, want orca-v1-", unitNamePrefix)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSystemdEmitter_TaskGroupTwoTasks(t *testing.T) {
|
||||
// P06: a task group with two tasks renders one unit per task plus
|
||||
// a grouping target unit. Each per-task unit is
|
||||
// `orca-v1-alloc-<alloc-id>-<task-name>.service`, carries
|
||||
// `PartOf=orca-v1-alloc-<alloc-id>.target`, and is
|
||||
// `WantedBy=multi-user.target`. The target unit lists every
|
||||
// per-task unit under Wants=.
|
||||
spec := &jobspec.WorkloadSpec{
|
||||
Kind: "Service",
|
||||
Name: "web",
|
||||
Tasks: []jobspec.TaskGroupTask{
|
||||
{
|
||||
Name: "app",
|
||||
Command: "/usr/bin/httpd -f",
|
||||
Runtime: &jobspec.RuntimeBlock{OneOf: "process"},
|
||||
},
|
||||
{
|
||||
Name: "sidecar",
|
||||
Command: "/bin/wasm-runner sidecar.wasm",
|
||||
Runtime: &jobspec.RuntimeBlock{OneOf: "wasm"},
|
||||
},
|
||||
},
|
||||
}
|
||||
files, err := SystemdEmitter{}.Render(spec, &Node{Hostname: "n1"})
|
||||
if err != nil {
|
||||
t.Fatalf("Render: %v", err)
|
||||
}
|
||||
// 2 per-task units + 1 target unit.
|
||||
if len(files) != 3 {
|
||||
t.Fatalf("got %d files, want 3 (2 per-task units + 1 target)", len(files))
|
||||
}
|
||||
wantApp := "/etc/systemd/system/orca-v1-alloc-web-app.service"
|
||||
wantSide := "/etc/systemd/system/orca-v1-alloc-web-sidecar.service"
|
||||
wantTarget := "/etc/systemd/system/orca-v1-alloc-web.target"
|
||||
paths := make(map[string]*File, len(files))
|
||||
for i := range files {
|
||||
paths[files[i].Path] = &files[i]
|
||||
}
|
||||
if _, ok := paths[wantApp]; !ok {
|
||||
t.Errorf("missing per-task unit %q; got paths %v", wantApp, filePaths(files))
|
||||
}
|
||||
if _, ok := paths[wantSide]; !ok {
|
||||
t.Errorf("missing per-task unit %q; got paths %v", wantSide, filePaths(files))
|
||||
}
|
||||
if _, ok := paths[wantTarget]; !ok {
|
||||
t.Errorf("missing target unit %q; got paths %v", wantTarget, filePaths(files))
|
||||
}
|
||||
if _, ok := paths[wantTarget]; !ok {
|
||||
t.Errorf("missing target unit %q; got paths %v", wantTarget, filePaths(files))
|
||||
}
|
||||
// Verify PartOf relations and ExecStart on per-task units.
|
||||
app := paths[wantApp]
|
||||
if !strings.Contains(app.Content, "PartOf=orca-v1-alloc-web.target") {
|
||||
t.Errorf("app unit missing PartOf=orca-v1-alloc-web.target\n%s", app.Content)
|
||||
}
|
||||
if !strings.Contains(app.Content, "ExecStart=/usr/bin/httpd -f") {
|
||||
t.Errorf("app unit missing ExecStart=/usr/bin/httpd -f\n%s", app.Content)
|
||||
}
|
||||
if !strings.Contains(app.Content, "WantedBy=multi-user.target") {
|
||||
t.Errorf("app unit missing WantedBy=multi-user.target\n%s", app.Content)
|
||||
}
|
||||
side := paths[wantSide]
|
||||
if !strings.Contains(side.Content, "PartOf=orca-v1-alloc-web.target") {
|
||||
t.Errorf("sidecar unit missing PartOf=orca-v1-alloc-web.target\n%s", side.Content)
|
||||
}
|
||||
if !strings.Contains(side.Content, "ExecStart=/bin/wasm-runner sidecar.wasm") {
|
||||
t.Errorf("sidecar unit missing ExecStart\n%s", side.Content)
|
||||
}
|
||||
// Verify the target unit Wants= both per-task units.
|
||||
target := paths[wantTarget]
|
||||
if !strings.Contains(target.Content, "Wants=orca-v1-alloc-web-app.service") {
|
||||
t.Errorf("target missing Wants=...app.service\n%s", target.Content)
|
||||
}
|
||||
if !strings.Contains(target.Content, "Wants=orca-v1-alloc-web-sidecar.service") {
|
||||
t.Errorf("target missing Wants=...sidecar.service\n%s", target.Content)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSystemdEmitter_TaskGroupInheritsTopLevelRuntime(t *testing.T) {
|
||||
// P06: a task that omits its own runtime inherits the top-level
|
||||
// spec.Runtime as the per-group default. The per-task unit's
|
||||
// ExecStart must come from the top-level runtime command when
|
||||
// the task has no own command and no own runtime.
|
||||
spec := &jobspec.WorkloadSpec{
|
||||
Kind: "Service",
|
||||
Name: "web",
|
||||
Runtime: &jobspec.RuntimeBlock{OneOf: "process", Command: "/bin/default"},
|
||||
Tasks: []jobspec.TaskGroupTask{
|
||||
{Name: "app"},
|
||||
{Name: "sidecar", Command: "/bin/override"},
|
||||
},
|
||||
}
|
||||
files, err := SystemdEmitter{}.Render(spec, &Node{})
|
||||
if err != nil {
|
||||
t.Fatalf("Render: %v", err)
|
||||
}
|
||||
if len(files) != 3 {
|
||||
t.Fatalf("got %d files, want 3", len(files))
|
||||
}
|
||||
appContent := findUnitContent(files, "/etc/systemd/system/orca-v1-alloc-web-app.service")
|
||||
if appContent == "" {
|
||||
t.Fatalf("missing app unit; paths %v", filePaths(files))
|
||||
}
|
||||
if !strings.Contains(appContent, "ExecStart=/bin/default") {
|
||||
t.Errorf("app unit should inherit top-level command /bin/default\n%s", appContent)
|
||||
}
|
||||
sideContent := findUnitContent(files, "/etc/systemd/system/orca-v1-alloc-web-sidecar.service")
|
||||
if sideContent == "" {
|
||||
t.Fatalf("missing sidecar unit; paths %v", filePaths(files))
|
||||
}
|
||||
if !strings.Contains(sideContent, "ExecStart=/bin/override") {
|
||||
t.Errorf("sidecar unit should use its own command /bin/override\n%s", sideContent)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSystemdEmitter_TaskGroupNoCommandError(t *testing.T) {
|
||||
// P06: a task with no resolvable command (no task.Command, no
|
||||
// task.Runtime, no top-level Runtime) is an error.
|
||||
spec := &jobspec.WorkloadSpec{
|
||||
Kind: "Service",
|
||||
Name: "web",
|
||||
Tasks: []jobspec.TaskGroupTask{{Name: "app"}},
|
||||
}
|
||||
_, err := SystemdEmitter{}.Render(spec, &Node{})
|
||||
if err == nil {
|
||||
t.Fatal("expected error for task with no runtime, got nil")
|
||||
}
|
||||
if !strings.Contains(err.Error(), "no runtime") {
|
||||
t.Errorf("error = %q, want 'no runtime'", err.Error())
|
||||
}
|
||||
}
|
||||
|
||||
func TestSystemdEmitter_TaskGroupEmptyCommandError(t *testing.T) {
|
||||
// P06: a task whose resolved runtime command is empty/whitespace
|
||||
// is an error (mirrors the single-process rule).
|
||||
spec := &jobspec.WorkloadSpec{
|
||||
Kind: "Service",
|
||||
Name: "web",
|
||||
Runtime: &jobspec.RuntimeBlock{OneOf: "process", Command: " "},
|
||||
Tasks: []jobspec.TaskGroupTask{{Name: "app"}},
|
||||
}
|
||||
_, err := SystemdEmitter{}.Render(spec, &Node{})
|
||||
if err == nil {
|
||||
t.Fatal("expected error for empty command, got nil")
|
||||
}
|
||||
if !strings.Contains(err.Error(), "command is empty") {
|
||||
t.Errorf("error = %q, want 'command is empty'", err.Error())
|
||||
}
|
||||
}
|
||||
|
||||
func TestSystemdEmitter_NoTasksBackwardCompat(t *testing.T) {
|
||||
// Backward compat: a spec with no Tasks renders exactly one unit
|
||||
// (the historical single-process shape).
|
||||
spec := &jobspec.WorkloadSpec{
|
||||
Kind: "Job",
|
||||
Name: "backup",
|
||||
Runtime: &jobspec.RuntimeBlock{OneOf: "process", Command: "/bin/rsync"},
|
||||
}
|
||||
files, err := SystemdEmitter{}.Render(spec, &Node{})
|
||||
if err != nil {
|
||||
t.Fatalf("Render: %v", err)
|
||||
}
|
||||
if len(files) != 1 {
|
||||
t.Fatalf("got %d files, want 1 (backward compat)", len(files))
|
||||
}
|
||||
if files[0].Path != "/etc/systemd/system/orca-v1-backup.service" {
|
||||
t.Errorf("Path = %q, want /etc/systemd/system/orca-v1-backup.service", files[0].Path)
|
||||
}
|
||||
}
|
||||
|
||||
func filePaths(files []File) []string {
|
||||
out := make([]string, len(files))
|
||||
for i, f := range files {
|
||||
out[i] = f.Path
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func findUnitContent(files []File, path string) string {
|
||||
for _, f := range files {
|
||||
if f.Path == path {
|
||||
return f.Content
|
||||
}
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
@@ -74,6 +74,27 @@ type WorkloadSpec struct {
|
||||
// Timeout is an optional execution timeout (duration string) for
|
||||
// Job. Populated by P04.
|
||||
Timeout string
|
||||
|
||||
// Tasks is the task-group list for multi-process services (P06,
|
||||
// PRD §9.1). When non-empty, the alloc runs one systemd unit per
|
||||
// task (`orca-v1-alloc-<alloc-id>-<task-name>.service`) all
|
||||
// grouped under a single `<alloc-id>.target`. When nil/empty,
|
||||
// the alloc is a single-process alloc driven by the top-level
|
||||
// Runtime block (backward compat). Tasks that omit their own
|
||||
// runtime inherit the top-level Runtime as the per-group default.
|
||||
Tasks []TaskGroupTask
|
||||
}
|
||||
|
||||
// TaskGroupTask is a single task within a task group (P06, PRD §9.1).
|
||||
// Each task has its own runtime (a wasm task + a process sidecar is
|
||||
// allowed), its own command, and an optional env overlay. When
|
||||
// Runtime is nil, the task inherits the top-level
|
||||
// WorkloadSpec.Runtime (the per-group default).
|
||||
type TaskGroupTask struct {
|
||||
Name string
|
||||
Runtime *RuntimeBlock
|
||||
Env map[string]string
|
||||
Command string
|
||||
}
|
||||
|
||||
// RuntimeBlock is a minimal runtime abstraction surface populated by the
|
||||
@@ -384,12 +405,18 @@ func parseFrontmatterBlock(block string) (*WorkloadSpec, error) {
|
||||
secLifecycle
|
||||
secAffinity
|
||||
secConstraints
|
||||
secTasks
|
||||
secTaskEnv
|
||||
secTaskRuntime
|
||||
)
|
||||
cur := secNone
|
||||
var curPort *PortSpec
|
||||
var curVol *VolumeSpec
|
||||
var curAffinity *AffinityRule
|
||||
var lifecycleCur string
|
||||
var curTask *TaskGroupTask
|
||||
var taskIndent int
|
||||
var taskFieldIndent int
|
||||
|
||||
flushPort := func() {
|
||||
if curPort != nil {
|
||||
@@ -409,6 +436,29 @@ func parseFrontmatterBlock(block string) (*WorkloadSpec, error) {
|
||||
curAffinity = nil
|
||||
}
|
||||
}
|
||||
flushTask := func() {
|
||||
if curTask != nil {
|
||||
spec.Tasks = append(spec.Tasks, *curTask)
|
||||
curTask = nil
|
||||
}
|
||||
}
|
||||
// taskSubBlock returns the sub-section to switch to when the
|
||||
// given `key: value` line opens a nested block under a task
|
||||
// (`env:` → secTaskEnv, `runtime:` → secTaskRuntime). Returns
|
||||
// secTasks for non-block keys (no switch).
|
||||
taskSubBlock := func(kvLine string) section {
|
||||
key, _, ok := splitKV(kvLine)
|
||||
if !ok {
|
||||
return secTasks
|
||||
}
|
||||
switch key {
|
||||
case "env":
|
||||
return secTaskEnv
|
||||
case "runtime":
|
||||
return secTaskRuntime
|
||||
}
|
||||
return secTasks
|
||||
}
|
||||
|
||||
for lineNo, raw := range lines {
|
||||
line := stripComment(raw)
|
||||
@@ -423,6 +473,7 @@ func parseFrontmatterBlock(block string) (*WorkloadSpec, error) {
|
||||
flushPort()
|
||||
flushVol()
|
||||
flushAffinity()
|
||||
flushTask()
|
||||
cur = secNone
|
||||
|
||||
key, val, ok := splitKV(trimmed)
|
||||
@@ -500,6 +551,10 @@ func parseFrontmatterBlock(block string) (*WorkloadSpec, error) {
|
||||
} else {
|
||||
cur = secAffinity
|
||||
}
|
||||
case "tasks":
|
||||
cur = secTasks
|
||||
taskIndent = -1
|
||||
taskFieldIndent = -1
|
||||
default:
|
||||
// Unknown top-level key are ignored (forward-compat).
|
||||
cur = secNone
|
||||
@@ -720,11 +775,119 @@ func parseFrontmatterBlock(block string) (*WorkloadSpec, error) {
|
||||
spec.Constraints = append(spec.Constraints, unquote(item))
|
||||
}
|
||||
}
|
||||
case secTasks:
|
||||
// Tasks is a list of task objects. A `- ` at the list
|
||||
// indent opens a new task; deeper-indented lines belong
|
||||
// to the current task's fields (name, command) or
|
||||
// nested sub-blocks (runtime, env).
|
||||
if strings.HasPrefix(trimmed, "- ") || trimmed == "-" {
|
||||
if taskIndent < 0 {
|
||||
taskIndent = indent
|
||||
taskFieldIndent = indent + 2
|
||||
}
|
||||
if indent == taskIndent {
|
||||
flushTask()
|
||||
t := TaskGroupTask{}
|
||||
curTask = &t
|
||||
rest := strings.TrimSpace(strings.TrimPrefix(trimmed, "-"))
|
||||
if rest != "" {
|
||||
if applyTaskKV(curTask, rest) {
|
||||
cur = taskSubBlock(rest)
|
||||
}
|
||||
}
|
||||
continue
|
||||
}
|
||||
}
|
||||
if curTask != nil {
|
||||
if applyTaskKV(curTask, trimmed) {
|
||||
cur = taskSubBlock(trimmed)
|
||||
}
|
||||
}
|
||||
case secTaskEnv:
|
||||
if curTask == nil {
|
||||
cur = secTasks
|
||||
continue
|
||||
}
|
||||
// Pop back to the task field level when the indent
|
||||
// returns to taskFieldIndent (the next sibling
|
||||
// field or a new `- ` list item). The line is then
|
||||
// reprocessed as a task field.
|
||||
if taskFieldIndent > 0 && indent <= taskFieldIndent {
|
||||
cur = secTasks
|
||||
if indent == taskIndent && (strings.HasPrefix(trimmed, "- ") || trimmed == "-") {
|
||||
flushTask()
|
||||
t := TaskGroupTask{}
|
||||
curTask = &t
|
||||
rest := strings.TrimSpace(strings.TrimPrefix(trimmed, "-"))
|
||||
if rest != "" {
|
||||
if applyTaskKV(curTask, rest) {
|
||||
cur = taskSubBlock(rest)
|
||||
}
|
||||
}
|
||||
continue
|
||||
}
|
||||
if applyTaskKV(curTask, trimmed) {
|
||||
cur = taskSubBlock(trimmed)
|
||||
}
|
||||
continue
|
||||
}
|
||||
if curTask.Env == nil {
|
||||
curTask.Env = map[string]string{}
|
||||
}
|
||||
key, val, ok := splitKV(trimmed)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
if val == "" {
|
||||
curTask.Env[key] = ""
|
||||
} else if strings.HasPrefix(val, "{") && strings.HasSuffix(val, "}") {
|
||||
curTask.Env[key] = val
|
||||
} else {
|
||||
curTask.Env[key] = unquote(val)
|
||||
}
|
||||
case secTaskRuntime:
|
||||
if curTask == nil || curTask.Runtime == nil {
|
||||
cur = secTasks
|
||||
continue
|
||||
}
|
||||
// Pop back to the task field level (see secTaskEnv).
|
||||
if taskFieldIndent > 0 && indent <= taskFieldIndent {
|
||||
cur = secTasks
|
||||
if indent == taskIndent && (strings.HasPrefix(trimmed, "- ") || trimmed == "-") {
|
||||
flushTask()
|
||||
t := TaskGroupTask{}
|
||||
curTask = &t
|
||||
rest := strings.TrimSpace(strings.TrimPrefix(trimmed, "-"))
|
||||
if rest != "" {
|
||||
if applyTaskKV(curTask, rest) {
|
||||
cur = taskSubBlock(rest)
|
||||
}
|
||||
}
|
||||
continue
|
||||
}
|
||||
if applyTaskKV(curTask, trimmed) {
|
||||
cur = taskSubBlock(trimmed)
|
||||
}
|
||||
continue
|
||||
}
|
||||
key, val, ok := splitKV(trimmed)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
switch key {
|
||||
case "one_of":
|
||||
curTask.Runtime.OneOf = unquote(val)
|
||||
case "image":
|
||||
curTask.Runtime.Image = unquote(val)
|
||||
case "command":
|
||||
curTask.Runtime.Command = unquote(val)
|
||||
}
|
||||
}
|
||||
}
|
||||
flushPort()
|
||||
flushVol()
|
||||
flushAffinity()
|
||||
flushTask()
|
||||
return spec, nil
|
||||
}
|
||||
|
||||
@@ -791,6 +954,34 @@ func applyAffinityKV(r *AffinityRule, s string) {
|
||||
}
|
||||
}
|
||||
|
||||
// applyTaskKV applies a `key: value` pair to the current TaskGroupTask.
|
||||
// The returned bool reports whether the key opened a nested sub-block
|
||||
// (`env` or `runtime`); when true the caller switches the parser
|
||||
// section to the corresponding sub-block handler.
|
||||
func applyTaskKV(t *TaskGroupTask, s string) (openedSubBlock bool) {
|
||||
key, val, ok := splitKV(s)
|
||||
if !ok {
|
||||
return false
|
||||
}
|
||||
switch key {
|
||||
case "name":
|
||||
t.Name = unquote(val)
|
||||
case "command":
|
||||
t.Command = unquote(val)
|
||||
case "env":
|
||||
if t.Env == nil {
|
||||
t.Env = map[string]string{}
|
||||
}
|
||||
return true
|
||||
case "runtime":
|
||||
if t.Runtime == nil {
|
||||
t.Runtime = &RuntimeBlock{}
|
||||
}
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// appendLifecycleCmd appends a command to the named lifecycle hook list
|
||||
// (pre_stop or post_start) on the given LifecycleBlock.
|
||||
func appendLifecycleCmd(lb *LifecycleBlock, name, cmd string) {
|
||||
|
||||
@@ -701,3 +701,152 @@ func TestParseMarkdown_FullServiceSpec(t *testing.T) {
|
||||
t.Errorf("Body = %q, want %q (R-015)", spec.Body, "# body\n")
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseMarkdown_TasksBlock(t *testing.T) {
|
||||
// P06: a task group with two tasks, each carrying its own runtime
|
||||
// and command. The parser must populate spec.Tasks with two
|
||||
// entries preserving name, runtime (one_of/image/command), and
|
||||
// the task-level command.
|
||||
input := "---\n" +
|
||||
"kind: Service\n" +
|
||||
"name: web\n" +
|
||||
"tasks:\n" +
|
||||
" - name: app\n" +
|
||||
" runtime:\n" +
|
||||
" one_of: process\n" +
|
||||
" image: docker.io/nginx:latest\n" +
|
||||
" command: /usr/bin/httpd -f\n" +
|
||||
" command: /usr/bin/httpd -f\n" +
|
||||
" - name: sidecar\n" +
|
||||
" runtime:\n" +
|
||||
" one_of: wasm\n" +
|
||||
" command: /bin/wasm-runner sidecar.wasm\n" +
|
||||
" command: /bin/wasm-runner sidecar.wasm\n" +
|
||||
"---\nbody\n"
|
||||
spec, err := ParseMarkdown([]byte(input))
|
||||
if err != nil {
|
||||
t.Fatalf("ParseMarkdown: %v", err)
|
||||
}
|
||||
if len(spec.Tasks) != 2 {
|
||||
t.Fatalf("Tasks = %d, want 2", len(spec.Tasks))
|
||||
}
|
||||
app := spec.Tasks[0]
|
||||
if app.Name != "app" {
|
||||
t.Errorf("Tasks[0].Name = %q, want app", app.Name)
|
||||
}
|
||||
if app.Runtime == nil {
|
||||
t.Fatal("Tasks[0].Runtime is nil")
|
||||
}
|
||||
if app.Runtime.OneOf != "process" {
|
||||
t.Errorf("Tasks[0].Runtime.OneOf = %q, want process", app.Runtime.OneOf)
|
||||
}
|
||||
if app.Runtime.Image != "docker.io/nginx:latest" {
|
||||
t.Errorf("Tasks[0].Runtime.Image = %q", app.Runtime.Image)
|
||||
}
|
||||
if app.Runtime.Command != "/usr/bin/httpd -f" {
|
||||
t.Errorf("Tasks[0].Runtime.Command = %q", app.Runtime.Command)
|
||||
}
|
||||
if app.Command != "/usr/bin/httpd -f" {
|
||||
t.Errorf("Tasks[0].Command = %q", app.Command)
|
||||
}
|
||||
side := spec.Tasks[1]
|
||||
if side.Name != "sidecar" {
|
||||
t.Errorf("Tasks[1].Name = %q, want sidecar", side.Name)
|
||||
}
|
||||
if side.Runtime == nil || side.Runtime.OneOf != "wasm" {
|
||||
t.Errorf("Tasks[1].Runtime = %+v, want one_of=wasm", side.Runtime)
|
||||
}
|
||||
if side.Command != "/bin/wasm-runner sidecar.wasm" {
|
||||
t.Errorf("Tasks[1].Command = %q", side.Command)
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseMarkdown_TasksBlockWithEnv(t *testing.T) {
|
||||
// P06: a task group task carrying an env overlay.
|
||||
input := "---\n" +
|
||||
"kind: Service\n" +
|
||||
"name: web\n" +
|
||||
"tasks:\n" +
|
||||
" - name: app\n" +
|
||||
" command: /usr/bin/httpd\n" +
|
||||
" env:\n" +
|
||||
" LOG_LEVEL: debug\n" +
|
||||
" REGION: us\n" +
|
||||
"---\nbody\n"
|
||||
spec, err := ParseMarkdown([]byte(input))
|
||||
if err != nil {
|
||||
t.Fatalf("ParseMarkdown: %v", err)
|
||||
}
|
||||
if len(spec.Tasks) != 1 {
|
||||
t.Fatalf("Tasks = %d, want 1", len(spec.Tasks))
|
||||
}
|
||||
task := spec.Tasks[0]
|
||||
if task.Env == nil {
|
||||
t.Fatal("Tasks[0].Env is nil")
|
||||
}
|
||||
if got := task.Env["LOG_LEVEL"]; got != "debug" {
|
||||
t.Errorf("Env[LOG_LEVEL] = %q, want debug", got)
|
||||
}
|
||||
if got := task.Env["REGION"]; got != "us" {
|
||||
t.Errorf("Env[REGION] = %q, want us", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseMarkdown_TasksBlockInheritsTopLevelRuntime(t *testing.T) {
|
||||
// P06: when a task omits its own runtime, the top-level runtime
|
||||
// is the per-group default. The parser must NOT create a task
|
||||
// runtime when the task block lacks a `runtime:` sub-block; the
|
||||
// emitter/validator resolve the default from spec.Runtime.
|
||||
input := "---\n" +
|
||||
"kind: Service\n" +
|
||||
"name: web\n" +
|
||||
"runtime:\n" +
|
||||
" one_of: process\n" +
|
||||
" command: /bin/default\n" +
|
||||
"tasks:\n" +
|
||||
" - name: app\n" +
|
||||
" command: /bin/app\n" +
|
||||
" - name: sidecar\n" +
|
||||
" command: /bin/sidecar\n" +
|
||||
"---\nbody\n"
|
||||
spec, err := ParseMarkdown([]byte(input))
|
||||
if err != nil {
|
||||
t.Fatalf("ParseMarkdown: %v", err)
|
||||
}
|
||||
if spec.Runtime == nil || spec.Runtime.OneOf != "process" {
|
||||
t.Fatalf("top-level runtime not parsed: %+v", spec.Runtime)
|
||||
}
|
||||
if len(spec.Tasks) != 2 {
|
||||
t.Fatalf("Tasks = %d, want 2", len(spec.Tasks))
|
||||
}
|
||||
for i, task := range spec.Tasks {
|
||||
if task.Runtime != nil {
|
||||
t.Errorf("Tasks[%d].Runtime should be nil (inherit top-level), got %+v", i, task.Runtime)
|
||||
}
|
||||
}
|
||||
if spec.Tasks[0].Name != "app" || spec.Tasks[1].Name != "sidecar" {
|
||||
t.Errorf("task names = %q, %q", spec.Tasks[0].Name, spec.Tasks[1].Name)
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseMarkdown_NoTasksBackwardCompat(t *testing.T) {
|
||||
// Backward compat: a spec with no `tasks:` block parses as a
|
||||
// single-process alloc; spec.Tasks must be empty/nil.
|
||||
input := "---\n" +
|
||||
"kind: Job\n" +
|
||||
"name: backup\n" +
|
||||
"runtime:\n" +
|
||||
" one_of: process\n" +
|
||||
" command: /bin/rsync\n" +
|
||||
"---\nbody\n"
|
||||
spec, err := ParseMarkdown([]byte(input))
|
||||
if err != nil {
|
||||
t.Fatalf("ParseMarkdown: %v", err)
|
||||
}
|
||||
if len(spec.Tasks) != 0 {
|
||||
t.Fatalf("Tasks = %d, want 0 (backward compat)", len(spec.Tasks))
|
||||
}
|
||||
if spec.Runtime == nil || spec.Runtime.Command != "/bin/rsync" {
|
||||
t.Errorf("Runtime = %+v, want command=/bin/rsync", spec.Runtime)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,43 @@
|
||||
# C-01 Grill Gate Evaluation — wasmtime / CGO
|
||||
|
||||
**Gate**: C-01 (wasmtime/CGO evaluation), gating P07b.
|
||||
|
||||
**Question**: The wasmtime Go binding
|
||||
(`github.com/bytecodealliance/wasmtime-go`) is CGO-based. Does adopting
|
||||
it revoke D-002 (modernc/sqlite CGO-free cross-compile story)?
|
||||
|
||||
## Evaluation
|
||||
|
||||
| Option | CGO required? | Cross-compile impact | Decision |
|
||||
|--------|---------------|----------------------|----------|
|
||||
| A. Use `bytecodealliance/wasmtime-go` (Go binding) | **YES** — the binding links libwasmtime via cgo | Revokes D-002 — Go cross-compile (`GOOS=linux GOARCH=arm64 go build`) breaks; CGO toolchain needed on every build host; static-binary story lost | REJECTED |
|
||||
| B. Use the `wasmtime` CLI (apt-installed on the peer) via SSH exec | **NO** — pure Go code, shells out to a CLI over SSH (same pattern as podman/qm/pct) | None — D-002 preserved | **ACCEPTED** |
|
||||
| C. Use an alternative pure-Go WASM runtime (e.g. wazero) | No CGO | Pure-Go alternative exists; but wazero's wasmtime-compat is incomplete (component model, WASI 0.2); different runtime semantics than the "wasmtime" operator surface promised in D-088 | DEFERRED (v0.10 evaluation if CLI-via-SSH proves insufficient) |
|
||||
|
||||
## Decision (full autonomy, auto-decision)
|
||||
|
||||
**Option B**: `WasmRuntime` uses the `wasmtime` CLI (apt-installed on
|
||||
the peer) via the SSH-push transport. It does NOT import
|
||||
`bytecodealliance/wasmtime-go` (or any other CGO package).
|
||||
|
||||
## C-01 Gate Status
|
||||
|
||||
**SATISFIED.** C-01 is satisfied:
|
||||
- wasmtime works without CGO (CLI-via-SSH pattern, identical to podman/qm/pct).
|
||||
- D-002 cross-compile story preserved (no CGO introduced anywhere in
|
||||
the runtime package or any orca Go code).
|
||||
- D-002 is NOT revoked.
|
||||
|
||||
## Recorded as D-187
|
||||
|
||||
See PROJECT.md v0.9 D-series: "wasmtime Go binding is CGO-based
|
||||
(bytecodealliance/wasmtime-go); orca uses the wasmtime CLI via SSH
|
||||
(apt-installed on peer) instead of the Go binding, avoiding CGO
|
||||
entirely. D-002 cross-compile story preserved. C-01 satisfied."
|
||||
|
||||
## Verification
|
||||
|
||||
- `go build ./internal/runtime/` succeeds with `CGO_ENABLED=0`.
|
||||
- `internal/runtime/wasm.go` imports only stdlib + sshpush (no
|
||||
wasmtime-go).
|
||||
- The full test suite (`go test ./...`) does not require CGO.
|
||||
@@ -0,0 +1,142 @@
|
||||
package runtime
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"strings"
|
||||
|
||||
"git.cloudinit.dev/coreci/orca/internal/sshpush"
|
||||
)
|
||||
|
||||
// PodmanRuntime implements Runtime for the "podman" one_of. It runs
|
||||
// `podman` on the peer over the SSH-push transport (P01). The
|
||||
// transport is injected via the constructor (dependency injection).
|
||||
//
|
||||
// Container lifecycle:
|
||||
//
|
||||
// - Prepare: `podman pull <image>`
|
||||
// - Start: `podman run -d --name orca-<alloc-id> <image> <command>`
|
||||
// - Stop: `podman stop <name>` then `podman rm <name>`
|
||||
// - Status: `podman inspect --format '{{.State.Running}}' <name>`
|
||||
//
|
||||
// The container name is `orca-<alloc-id>` (sanitized to lowercase +
|
||||
// alnum). The runtime keeps no in-process state — each call is a fresh
|
||||
// SSH exec against the peer.
|
||||
type PodmanRuntime struct {
|
||||
transport *sshpush.Transport
|
||||
}
|
||||
|
||||
// NewPodmanRuntime returns a PodmanRuntime backed by the given transport.
|
||||
func NewPodmanRuntime(t *sshpush.Transport) *PodmanRuntime {
|
||||
return &PodmanRuntime{transport: t}
|
||||
}
|
||||
|
||||
// containerName returns the deterministic container name for an alloc.
|
||||
func containerName(alloc *Alloc) string {
|
||||
id := strings.ToLower(alloc.ID)
|
||||
id = strings.Map(func(r rune) rune {
|
||||
if r >= 'a' && r <= 'z' || r >= '0' && r <= '9' || r == '-' || r == '_' {
|
||||
return r
|
||||
}
|
||||
return '-'
|
||||
}, id)
|
||||
return "orca-" + id
|
||||
}
|
||||
|
||||
// Prepare pulls the image on the peer.
|
||||
func (p *PodmanRuntime) Prepare(ctx context.Context, alloc *Alloc) error {
|
||||
image, err := imageFor(alloc)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
cmd := fmt.Sprintf("podman pull %q", image)
|
||||
if _, err := p.transport.Exec(ctx, alloc.Node, cmd); err != nil {
|
||||
return fmt.Errorf("podman: pull: %w", err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// Start runs `podman run -d --name <name> <image> <command>`.
|
||||
func (p *PodmanRuntime) Start(ctx context.Context, alloc *Alloc) (int, error) {
|
||||
image, err := imageFor(alloc)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
cmdStr, _ := commandFor(alloc)
|
||||
name := containerName(alloc)
|
||||
cmd := fmt.Sprintf("podman run -d --name %s %q %s", name, image, cmdStr)
|
||||
out, err := p.transport.Exec(ctx, alloc.Node, cmd)
|
||||
if err != nil {
|
||||
return 0, fmt.Errorf("podman: run: %w", err)
|
||||
}
|
||||
// The container ID is the first 12 chars of the printed hash. We
|
||||
// don't keep it — Stop/Status use the name — but return a stable
|
||||
// synthetic PID derived from the first 4 bytes of the hash for
|
||||
// the interface contract.
|
||||
cid := strings.TrimSpace(string(out))
|
||||
return podmanCidToPID(cid), nil
|
||||
}
|
||||
|
||||
// Stop stops and removes the container.
|
||||
func (p *PodmanRuntime) Stop(ctx context.Context, alloc *Alloc) error {
|
||||
name := containerName(alloc)
|
||||
if _, err := p.transport.Exec(ctx, alloc.Node, fmt.Sprintf("podman stop %s", name)); err != nil {
|
||||
return fmt.Errorf("podman: stop: %w", err)
|
||||
}
|
||||
if _, err := p.transport.Exec(ctx, alloc.Node, fmt.Sprintf("podman rm %s", name)); err != nil {
|
||||
return fmt.Errorf("podman: rm: %w", err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// Status inspects the container's running state.
|
||||
func (p *PodmanRuntime) Status(ctx context.Context, alloc *Alloc) (State, error) {
|
||||
name := containerName(alloc)
|
||||
cmd := fmt.Sprintf("podman inspect --format '{{.State.Running}}' %s", name)
|
||||
out, err := p.transport.Exec(ctx, alloc.Node, cmd)
|
||||
if err != nil {
|
||||
return StateFailed, fmt.Errorf("podman: inspect: %w", err)
|
||||
}
|
||||
v := strings.TrimSpace(string(out))
|
||||
switch v {
|
||||
case "true":
|
||||
return StateRunning, nil
|
||||
case "false":
|
||||
return StateStopped, nil
|
||||
default:
|
||||
return StateFailed, fmt.Errorf("podman: unexpected inspect output %q", v)
|
||||
}
|
||||
}
|
||||
|
||||
// podmanCidToPID converts a container ID (hex hash) to a positive int
|
||||
// PID for the interface contract. It reads up to 4 hex chars.
|
||||
func podmanCidToPID(cid string) int {
|
||||
if len(cid) < 1 {
|
||||
return 1
|
||||
}
|
||||
n := len(cid)
|
||||
if n > 4 {
|
||||
n = 4
|
||||
}
|
||||
var pid int
|
||||
for i := 0; i < n; i++ {
|
||||
c := cid[i]
|
||||
pid = (pid << 4) | int(hexVal(c))
|
||||
}
|
||||
if pid <= 0 {
|
||||
pid = 1
|
||||
}
|
||||
return pid
|
||||
}
|
||||
|
||||
func hexVal(c byte) byte {
|
||||
switch {
|
||||
case c >= '0' && c <= '9':
|
||||
return c - '0'
|
||||
case c >= 'a' && c <= 'f':
|
||||
return c - 'a' + 10
|
||||
case c >= 'A' && c <= 'F':
|
||||
return c - 'A' + 10
|
||||
}
|
||||
return 0
|
||||
}
|
||||
@@ -0,0 +1,229 @@
|
||||
package runtime
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"git.cloudinit.dev/coreci/orca/internal/jobspec"
|
||||
"git.cloudinit.dev/coreci/orca/internal/sshpush"
|
||||
)
|
||||
|
||||
// allocNoImage returns an alloc whose Spec has a Runtime block with a
|
||||
// command but no image.
|
||||
func allocNoImage(runtime string) *Alloc {
|
||||
return &Alloc{
|
||||
ID: "x",
|
||||
Runtime: runtime,
|
||||
Spec: &jobspec.WorkloadSpec{
|
||||
Runtime: &jobspec.RuntimeBlock{Command: "/bin/true"},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// TestPodmanRuntime_HappyPath wires a fake server that responds to
|
||||
// podman pull/run/stop/rm/inspect and verifies the full lifecycle.
|
||||
func TestPodmanRuntime_HappyPath(t *testing.T) {
|
||||
srv := newFakeServer(t)
|
||||
defer srv.close()
|
||||
|
||||
const cid = "abc123def456"
|
||||
srv.setHandler("podman pull", func(cmd string) ([]byte, int) { return nil, 0 })
|
||||
srv.setHandler("podman run", func(cmd string) ([]byte, int) { return []byte(cid + "\n"), 0 })
|
||||
srv.setHandler("podman stop", func(cmd string) ([]byte, int) { return nil, 0 })
|
||||
srv.setHandler("podman rm", func(cmd string) ([]byte, int) { return nil, 0 })
|
||||
srv.setHandler("podman inspect", func(cmd string) ([]byte, int) {
|
||||
return []byte("true\n"), 0
|
||||
})
|
||||
|
||||
tr := realTransport(t, srv)
|
||||
defer tr.Close()
|
||||
|
||||
p := NewPodmanRuntime(tr)
|
||||
a := allocWithNode("podman", "docker.io/library/alpine:latest", "sleep 30", srv.addr())
|
||||
|
||||
ctx, cancel := withTimeout(10 * time.Second)
|
||||
defer cancel()
|
||||
if err := p.Prepare(ctx, a); err != nil {
|
||||
t.Fatalf("Prepare: %v", err)
|
||||
}
|
||||
pid, err := p.Start(ctx, a)
|
||||
if err != nil {
|
||||
t.Fatalf("Start: %v", err)
|
||||
}
|
||||
if pid <= 0 {
|
||||
t.Fatalf("pid = %d, want > 0", pid)
|
||||
}
|
||||
st, err := p.Status(ctx, a)
|
||||
if err != nil {
|
||||
t.Fatalf("Status: %v", err)
|
||||
}
|
||||
if st != StateRunning {
|
||||
t.Errorf("Status = %q, want running", st)
|
||||
}
|
||||
if err := p.Stop(ctx, a); err != nil {
|
||||
t.Fatalf("Stop: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestPodmanRuntime_StatusFalse verifies Status returns stopped when
|
||||
// the container reports running=false.
|
||||
func TestPodmanRuntime_StatusFalse(t *testing.T) {
|
||||
srv := newFakeServer(t)
|
||||
defer srv.close()
|
||||
srv.setHandler("podman inspect", func(cmd string) ([]byte, int) {
|
||||
return []byte("false\n"), 0
|
||||
})
|
||||
tr := realTransport(t, srv)
|
||||
defer tr.Close()
|
||||
p := NewPodmanRuntime(tr)
|
||||
a := allocWithNode("podman", "img", "sleep 1", srv.addr())
|
||||
st, err := p.Status(context.Background(), a)
|
||||
if err != nil {
|
||||
t.Fatalf("Status: %v", err)
|
||||
}
|
||||
if st != StateStopped {
|
||||
t.Errorf("Status = %q, want stopped", st)
|
||||
}
|
||||
}
|
||||
|
||||
// TestPodmanRuntime_StatusBadOutput verifies Status returns failed on
|
||||
// unexpected inspect output.
|
||||
func TestPodmanRuntime_StatusBadOutput(t *testing.T) {
|
||||
srv := newFakeServer(t)
|
||||
defer srv.close()
|
||||
srv.setHandler("podman inspect", func(cmd string) ([]byte, int) {
|
||||
return []byte("garbage\n"), 0
|
||||
})
|
||||
tr := realTransport(t, srv)
|
||||
defer tr.Close()
|
||||
p := NewPodmanRuntime(tr)
|
||||
a := allocWithNode("podman", "img", "sleep 1", srv.addr())
|
||||
if _, err := p.Status(context.Background(), a); err == nil {
|
||||
t.Error("Status with bad output should error")
|
||||
}
|
||||
}
|
||||
|
||||
// TestPodmanRuntime_PrepareNoImage verifies Prepare errors when the
|
||||
// alloc has no image.
|
||||
func TestPodmanRuntime_PrepareNoImage(t *testing.T) {
|
||||
p := NewPodmanRuntime(nil)
|
||||
a := allocNoImage("podman")
|
||||
if err := p.Prepare(context.Background(), a); err == nil {
|
||||
t.Error("Prepare with no image should error")
|
||||
}
|
||||
}
|
||||
|
||||
// TestPodmanRuntime_PrepareTransportError verifies Prepare propagates a
|
||||
// transport error (podman pull fails).
|
||||
func TestPodmanRuntime_PrepareTransportError(t *testing.T) {
|
||||
srv := newFakeServer(t)
|
||||
defer srv.close()
|
||||
srv.setHandler("podman pull", func(cmd string) ([]byte, int) {
|
||||
return []byte("manifest unknown\n"), 2
|
||||
})
|
||||
tr := realTransport(t, srv)
|
||||
defer tr.Close()
|
||||
p := NewPodmanRuntime(tr)
|
||||
a := allocWithNode("podman", "img", "sleep 1", srv.addr())
|
||||
if err := p.Prepare(context.Background(), a); err == nil {
|
||||
t.Error("Prepare with failed pull should error")
|
||||
}
|
||||
}
|
||||
|
||||
// TestPodmanRuntime_StartNoImage verifies Start errors with no image.
|
||||
func TestPodmanRuntime_StartNoImage(t *testing.T) {
|
||||
p := NewPodmanRuntime(nil)
|
||||
a := allocNoImage("podman")
|
||||
if _, err := p.Start(context.Background(), a); err == nil {
|
||||
t.Error("Start with no image should error")
|
||||
}
|
||||
}
|
||||
|
||||
// TestPodmanRuntime_StopTransportError verifies Stop propagates errors.
|
||||
func TestPodmanRuntime_StopTransportError(t *testing.T) {
|
||||
srv := newFakeServer(t)
|
||||
defer srv.close()
|
||||
srv.setHandler("podman stop", func(cmd string) ([]byte, int) {
|
||||
return []byte("no such container\n"), 1
|
||||
})
|
||||
tr := realTransport(t, srv)
|
||||
defer tr.Close()
|
||||
p := NewPodmanRuntime(tr)
|
||||
a := allocWithNode("podman", "img", "sleep 1", srv.addr())
|
||||
if err := p.Stop(context.Background(), a); err == nil {
|
||||
t.Error("Stop with missing container should error")
|
||||
}
|
||||
}
|
||||
|
||||
// TestPodmanRuntime_StatusInspectError verifies Status returns failed
|
||||
// when inspect itself errors.
|
||||
func TestPodmanRuntime_StatusInspectError(t *testing.T) {
|
||||
srv := newFakeServer(t)
|
||||
defer srv.close()
|
||||
srv.setHandler("podman inspect", func(cmd string) ([]byte, int) {
|
||||
return []byte("no such container\n"), 1
|
||||
})
|
||||
tr := realTransport(t, srv)
|
||||
defer tr.Close()
|
||||
p := NewPodmanRuntime(tr)
|
||||
a := allocWithNode("podman", "img", "sleep 1", srv.addr())
|
||||
st, err := p.Status(context.Background(), a)
|
||||
if err == nil {
|
||||
t.Error("Status with inspect error should error")
|
||||
}
|
||||
if st != StateFailed {
|
||||
t.Errorf("Status = %q, want failed", st)
|
||||
}
|
||||
}
|
||||
|
||||
// TestPodmanCidToPID verifies the synthetic PID derivation.
|
||||
func TestPodmanCidToPID(t *testing.T) {
|
||||
if got := podmanCidToPID(""); got != 1 {
|
||||
t.Errorf("empty cid -> %d, want 1", got)
|
||||
}
|
||||
if got := podmanCidToPID("a"); got <= 0 {
|
||||
t.Errorf("single hex -> %d, want > 0", got)
|
||||
}
|
||||
if got := podmanCidToPID("abcd"); got <= 0 {
|
||||
t.Errorf("abcd -> %d, want > 0", got)
|
||||
}
|
||||
// non-hex chars fall through to 0 contributions but still yield
|
||||
// a positive result (>= 1 by the floor).
|
||||
if got := podmanCidToPID("xyz123"); got <= 0 {
|
||||
t.Errorf("xyz123 -> %d, want > 0", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestContainerNameSanitization verifies the container-name sanitizer
|
||||
// uppercases and strips disallowed characters.
|
||||
func TestContainerNameSanitization(t *testing.T) {
|
||||
a := &Alloc{ID: "ALLOC_1.2.3", Spec: nil, Runtime: "podman"}
|
||||
got := containerName(a)
|
||||
if !strings.HasPrefix(got, "orca-") {
|
||||
t.Errorf("containerName = %q, want orca- prefix", got)
|
||||
}
|
||||
if strings.Contains(got, ".") {
|
||||
t.Errorf("containerName = %q, should not contain '.'", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestPodmanRuntime_DialError verifies Prepare fails fast when the
|
||||
// peer is unreachable (no fake server).
|
||||
func TestPodmanRuntime_DialError(t *testing.T) {
|
||||
srv := newFakeServer(t)
|
||||
// close immediately so dial fails.
|
||||
srv.close()
|
||||
tr := realTransport(t, srv)
|
||||
defer tr.Close()
|
||||
p := NewPodmanRuntime(tr)
|
||||
a := allocWithNode("podman", "img", "sleep 1", srv.addr())
|
||||
if err := p.Prepare(context.Background(), a); err == nil {
|
||||
t.Error("Prepare against dead peer should error")
|
||||
} else if !errors.Is(err, sshpush.ErrTransient) && !errors.Is(err, sshpush.ErrPermanent) {
|
||||
// acceptable: either transient (retry exhausted) or permanent.
|
||||
t.Logf("Prepare err (acceptable): %v", err)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,204 @@
|
||||
package runtime
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"os"
|
||||
"os/exec"
|
||||
"strconv"
|
||||
"strings"
|
||||
"sync"
|
||||
"syscall"
|
||||
"time"
|
||||
)
|
||||
|
||||
// StopGrace is the default grace period between SIGTERM and SIGKILL
|
||||
// for ProcessRuntime.Stop (10s, matching the systemd default TimeoutStopSec).
|
||||
const StopGrace = 10 * time.Second
|
||||
|
||||
// ProcessRuntime implements Runtime for the "process" one_of using
|
||||
// os/exec. It is the in-process equivalent of the systemd unit the CLI
|
||||
// emits in production: the CLI emits a .service file and the peer's
|
||||
// systemd runs the process; ProcessRuntime starts the process directly
|
||||
// in the current Go process. It is intended for LOCAL testing and
|
||||
// hermetic CI — NOT for production (production uses the systemd emitter
|
||||
// + the peer's systemd, not this in-process path).
|
||||
//
|
||||
// It tracks started PIDs in an in-memory map; restarts of the CLI lose
|
||||
// that state (acceptable for the local-test use case).
|
||||
type ProcessRuntime struct {
|
||||
mu sync.Mutex
|
||||
pids map[string]int // alloc.ID -> PID
|
||||
procs map[int]*os.Process // PID -> process handle
|
||||
stopped map[string]bool // alloc.ID -> reported stopped after Stop
|
||||
}
|
||||
|
||||
// NewProcessRuntime returns a ProcessRuntime.
|
||||
func NewProcessRuntime() *ProcessRuntime {
|
||||
return &ProcessRuntime{
|
||||
pids: make(map[string]int),
|
||||
procs: make(map[int]*os.Process),
|
||||
stopped: make(map[string]bool),
|
||||
}
|
||||
}
|
||||
|
||||
// Prepare is a no-op for the process runtime: the systemd unit is
|
||||
// emitted by the systemd emitter (internal/emitter), not by the runtime.
|
||||
func (p *ProcessRuntime) Prepare(ctx context.Context, alloc *Alloc) error {
|
||||
_ = ctx
|
||||
_ = alloc
|
||||
return nil
|
||||
}
|
||||
|
||||
// Start execs the alloc's command and returns the PID. The process is
|
||||
// left running in the background; Stop terminates it.
|
||||
func (p *ProcessRuntime) Start(ctx context.Context, alloc *Alloc) (int, error) {
|
||||
cmdStr, err := commandFor(alloc)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
// Parse the command string into argv. A leading "exec" form
|
||||
// (shell-style) is NOT supported — the command must be a direct
|
||||
// argv[0] + args. Split on whitespace (simple, matches the existing
|
||||
// executor.go behaviour which takes Command + Args separately).
|
||||
parts := strings.Fields(cmdStr)
|
||||
if len(parts) == 0 {
|
||||
return 0, fmt.Errorf("process: empty command for alloc %s", alloc.ID)
|
||||
}
|
||||
|
||||
// Use a detached context so the process survives the request
|
||||
// context cancellation (the request ends; the workload keeps
|
||||
// running until Stop). We apply our own timeout for Start only.
|
||||
startCtx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
|
||||
defer cancel()
|
||||
|
||||
cmd := exec.CommandContext(startCtx, parts[0], parts[1:]...)
|
||||
// Detach the child from the parent's process group so it survives.
|
||||
cmd.SysProcAttr = &syscall.SysProcAttr{Setpgid: true}
|
||||
// Discard output for the runtime; the systemd unit captures logs
|
||||
// in production. For tests, callers that need output run their
|
||||
// own exec.Command.
|
||||
cmd.Stdout = os.Stdout
|
||||
cmd.Stderr = os.Stderr
|
||||
if err := cmd.Start(); err != nil {
|
||||
return 0, fmt.Errorf("process: start: %w", err)
|
||||
}
|
||||
pid := cmd.Process.Pid
|
||||
|
||||
// Background-reap the process so it doesn't become a zombie; we
|
||||
// only need the PID for Stop/Status. When the process exits
|
||||
// naturally, mark the alloc as stopped.
|
||||
go func() {
|
||||
_ = cmd.Wait()
|
||||
p.mu.Lock()
|
||||
delete(p.procs, pid)
|
||||
// Only mark stopped if the alloc is still associated with
|
||||
// this PID (Stop may have already removed the mapping).
|
||||
if cur, ok := p.pids[alloc.ID]; ok && cur == pid {
|
||||
delete(p.pids, alloc.ID)
|
||||
p.stopped[alloc.ID] = true
|
||||
}
|
||||
p.mu.Unlock()
|
||||
}()
|
||||
|
||||
p.mu.Lock()
|
||||
p.pids[alloc.ID] = pid
|
||||
p.procs[pid] = cmd.Process
|
||||
delete(p.stopped, alloc.ID)
|
||||
p.mu.Unlock()
|
||||
|
||||
return pid, nil
|
||||
}
|
||||
|
||||
// Stop sends SIGTERM, waits the grace period, then SIGKILL.
|
||||
func (p *ProcessRuntime) Stop(ctx context.Context, alloc *Alloc) error {
|
||||
p.mu.Lock()
|
||||
pid, ok := p.pids[alloc.ID]
|
||||
proc := p.procs[pid]
|
||||
p.mu.Unlock()
|
||||
if !ok || proc == nil {
|
||||
return nil // not running; idempotent
|
||||
}
|
||||
// SIGTERM the process group (negative PID).
|
||||
_ = syscall.Kill(-pid, syscall.SIGTERM)
|
||||
|
||||
grace := StopGrace
|
||||
if dl, ok := ctx.Deadline(); ok {
|
||||
if remaining := time.Until(dl); remaining > 0 && remaining < grace {
|
||||
grace = remaining
|
||||
}
|
||||
}
|
||||
deadline := time.Now().Add(grace)
|
||||
for time.Now().Before(deadline) {
|
||||
if !p.alive(pid) {
|
||||
p.forget(alloc.ID, pid)
|
||||
return nil
|
||||
}
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
p.forget(alloc.ID, pid)
|
||||
return ctx.Err()
|
||||
case <-time.After(100 * time.Millisecond):
|
||||
}
|
||||
}
|
||||
// SIGKILL the group.
|
||||
_ = syscall.Kill(-pid, syscall.SIGKILL)
|
||||
p.forget(alloc.ID, pid)
|
||||
return nil
|
||||
}
|
||||
|
||||
// Status reports the alloc's state by checking if the process is alive.
|
||||
func (p *ProcessRuntime) Status(ctx context.Context, alloc *Alloc) (State, error) {
|
||||
_ = ctx
|
||||
p.mu.Lock()
|
||||
pid, ok := p.pids[alloc.ID]
|
||||
wasStopped := p.stopped[alloc.ID]
|
||||
p.mu.Unlock()
|
||||
if !ok {
|
||||
if wasStopped {
|
||||
return StateStopped, nil
|
||||
}
|
||||
return StatePending, nil
|
||||
}
|
||||
if !p.alive(pid) {
|
||||
// Process exited but the reaper hasn't run yet; mark it
|
||||
// stopped and clean up.
|
||||
p.forget(alloc.ID, pid)
|
||||
return StateStopped, nil
|
||||
}
|
||||
return StateRunning, nil
|
||||
}
|
||||
|
||||
// alive reports whether the process with the given PID is still running.
|
||||
func (p *ProcessRuntime) alive(pid int) bool {
|
||||
proc, err := os.FindProcess(pid)
|
||||
if err != nil {
|
||||
return false
|
||||
}
|
||||
if err := proc.Signal(syscall.Signal(0)); err != nil {
|
||||
// ESRCH means the process is gone.
|
||||
return false
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
// forget removes the alloc/PID mapping and marks the alloc stopped.
|
||||
func (p *ProcessRuntime) forget(allocID string, pid int) {
|
||||
p.mu.Lock()
|
||||
delete(p.pids, allocID)
|
||||
delete(p.procs, pid)
|
||||
p.stopped[allocID] = true
|
||||
p.mu.Unlock()
|
||||
}
|
||||
|
||||
// PID returns the recorded PID for alloc (for tests/inspection).
|
||||
func (p *ProcessRuntime) PID(allocID string) (int, error) {
|
||||
p.mu.Lock()
|
||||
defer p.mu.Unlock()
|
||||
pid, ok := p.pids[allocID]
|
||||
if !ok {
|
||||
return 0, errors.New("process: no pid for alloc " + strconv.Quote(allocID))
|
||||
}
|
||||
return pid, nil
|
||||
}
|
||||
@@ -0,0 +1,172 @@
|
||||
package runtime
|
||||
|
||||
import (
|
||||
"context"
|
||||
"runtime"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"git.cloudinit.dev/coreci/orca/internal/jobspec"
|
||||
)
|
||||
|
||||
// TestProcessRuntime_PrepareIsNoop verifies Prepare is a no-op.
|
||||
func TestProcessRuntime_PrepareIsNoop(t *testing.T) {
|
||||
p := NewProcessRuntime()
|
||||
a := alloc("process", "", "/bin/true")
|
||||
if err := p.Prepare(context.Background(), a); err != nil {
|
||||
t.Fatalf("Prepare: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestProcessRuntime_PrepareNilSpec exercises the nil-spec branch
|
||||
// indirectly — Prepare is a no-op regardless of input.
|
||||
func TestProcessRuntime_PrepareNilSpec(t *testing.T) {
|
||||
p := NewProcessRuntime()
|
||||
if err := p.Prepare(context.Background(), &Alloc{ID: "x"}); err != nil {
|
||||
t.Fatalf("Prepare (nil spec) should still be no-op: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestProcessRuntime_StartStopStatus runs a real long-lived process
|
||||
// (sleep) and verifies Start -> Status(running) -> Stop -> Status(stopped).
|
||||
func TestProcessRuntime_StartStopStatus(t *testing.T) {
|
||||
if _, err := sleepBin(); err != nil {
|
||||
t.Skipf("sleep binary not available: %v", err)
|
||||
}
|
||||
p := NewProcessRuntime()
|
||||
sleepCmd, _ := sleepBin()
|
||||
a := alloc("process", "", sleepCmd+" 30")
|
||||
|
||||
ctx, cancel := withTimeout(10 * time.Second)
|
||||
defer cancel()
|
||||
pid, err := p.Start(ctx, a)
|
||||
if err != nil {
|
||||
t.Fatalf("Start: %v", err)
|
||||
}
|
||||
if pid <= 0 {
|
||||
t.Fatalf("pid = %d, want > 0", pid)
|
||||
}
|
||||
|
||||
st, err := p.Status(context.Background(), a)
|
||||
if err != nil {
|
||||
t.Fatalf("Status: %v", err)
|
||||
}
|
||||
if st != StateRunning {
|
||||
t.Errorf("Status = %q, want running", st)
|
||||
}
|
||||
|
||||
// Verify PID lookup.
|
||||
got, err := p.PID(a.ID)
|
||||
if err != nil || got != pid {
|
||||
t.Errorf("PID = %d/%v, want %d", got, err, pid)
|
||||
}
|
||||
|
||||
stopCtx, cancelStop := withTimeout(15 * time.Second)
|
||||
defer cancelStop()
|
||||
if err := p.Stop(stopCtx, a); err != nil {
|
||||
t.Fatalf("Stop: %v", err)
|
||||
}
|
||||
|
||||
st2, _ := p.Status(context.Background(), a)
|
||||
if st2 != StateStopped {
|
||||
t.Errorf("Status after Stop = %q, want stopped", st2)
|
||||
}
|
||||
}
|
||||
|
||||
// TestProcessRuntime_StopNotStarted verifies Stop is idempotent on a
|
||||
// never-started alloc.
|
||||
func TestProcessRuntime_StopNotStarted(t *testing.T) {
|
||||
p := NewProcessRuntime()
|
||||
a := alloc("process", "", "/bin/true")
|
||||
ctx, cancel := withTimeout(2 * time.Second)
|
||||
defer cancel()
|
||||
if err := p.Stop(ctx, a); err != nil {
|
||||
t.Errorf("Stop on never-started alloc should be no-op, got %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestProcessRuntime_StatusPending verifies Status returns pending for
|
||||
// an alloc that was never started.
|
||||
func TestProcessRuntime_StatusPending(t *testing.T) {
|
||||
p := NewProcessRuntime()
|
||||
a := alloc("process", "", "/bin/true")
|
||||
st, err := p.Status(context.Background(), a)
|
||||
if err != nil {
|
||||
t.Fatalf("Status: %v", err)
|
||||
}
|
||||
if st != StatePending {
|
||||
t.Errorf("Status = %q, want pending", st)
|
||||
}
|
||||
}
|
||||
|
||||
// TestProcessRuntime_StartNoCommand verifies Start errors on missing
|
||||
// command.
|
||||
func TestProcessRuntime_StartNoCommand(t *testing.T) {
|
||||
p := NewProcessRuntime()
|
||||
a := &Alloc{ID: "x", Spec: nil, Runtime: "process"}
|
||||
if _, err := p.Start(context.Background(), a); err == nil {
|
||||
t.Error("Start with nil spec should error")
|
||||
}
|
||||
}
|
||||
|
||||
// TestProcessRuntime_StartEmptyCommand verifies Start errors when
|
||||
// the command parses to zero argv.
|
||||
func TestProcessRuntime_StartEmptyCommand(t *testing.T) {
|
||||
p := NewProcessRuntime()
|
||||
a := &Alloc{
|
||||
ID: "e",
|
||||
Runtime: "process",
|
||||
Spec: &jobspec.WorkloadSpec{
|
||||
Runtime: &jobspec.RuntimeBlock{Command: " "},
|
||||
},
|
||||
}
|
||||
_, err := p.Start(context.Background(), a)
|
||||
if err == nil {
|
||||
t.Error("Start with empty command should error")
|
||||
}
|
||||
}
|
||||
|
||||
// TestProcessRuntime_StartBadBinary verifies Start propagates exec
|
||||
// errors for a missing binary.
|
||||
func TestProcessRuntime_StartBadBinary(t *testing.T) {
|
||||
p := NewProcessRuntime()
|
||||
a := alloc("process", "", "/no/such/binary/here")
|
||||
_, err := p.Start(context.Background(), a)
|
||||
if err == nil {
|
||||
t.Error("Start with missing binary should error")
|
||||
}
|
||||
}
|
||||
|
||||
// TestProcessRuntime_StopOnFinishedProcess verifies Stop on a process
|
||||
// that already exited (e.g. /bin/true) is a no-op (no error).
|
||||
func TestProcessRuntime_StopOnFinishedProcess(t *testing.T) {
|
||||
p := NewProcessRuntime()
|
||||
a := alloc("process", "", "/bin/true")
|
||||
_, _ = p.Start(context.Background(), a)
|
||||
// Give /bin/true time to exit.
|
||||
time.Sleep(200 * time.Millisecond)
|
||||
ctx, cancel := withTimeout(5 * time.Second)
|
||||
defer cancel()
|
||||
if err := p.Stop(ctx, a); err != nil {
|
||||
t.Errorf("Stop on exited process should be no-op, got %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestProcessRuntime_PIDUnknown verifies PID lookup errors for an
|
||||
// unknown alloc.
|
||||
func TestProcessRuntime_PIDUnknown(t *testing.T) {
|
||||
p := NewProcessRuntime()
|
||||
if _, err := p.PID("nope"); err == nil {
|
||||
t.Error("PID unknown should error")
|
||||
}
|
||||
}
|
||||
|
||||
// sleepBin returns the sleep command path ("sleep") on this OS.
|
||||
func sleepBin() (string, error) {
|
||||
// /bin/sleep exists on Linux; on other platforms fall back to
|
||||
// "sleep" (resolved via PATH).
|
||||
if runtime.GOOS == "linux" {
|
||||
return "/bin/sleep", nil
|
||||
}
|
||||
return "sleep", nil
|
||||
}
|
||||
@@ -0,0 +1,177 @@
|
||||
package runtime
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"strings"
|
||||
|
||||
"git.cloudinit.dev/coreci/orca/internal/sshpush"
|
||||
)
|
||||
|
||||
// vmidFor returns a deterministic 5-digit VMID derived from the alloc
|
||||
// ID (hash(alloc.ID) % 99999 + 1). Used by both pveVMRuntime and
|
||||
// pveCTRuntime — VM and container IDs share the same numeric space on
|
||||
// a Proxmox node, but the namespace is sparse (one alloc = one VMID)
|
||||
// so collisions are rare in practice.
|
||||
func vmidFor(alloc *Alloc) int {
|
||||
id := allocIDHash(alloc.ID)
|
||||
if id < 100 {
|
||||
id += 100
|
||||
}
|
||||
return id
|
||||
}
|
||||
|
||||
// PveVMRuntime implements Runtime for the "pve-vm" one_of using the
|
||||
// `qm` tool over SSH on a Proxmox peer (extends REQ-076). The transport
|
||||
// is injected.
|
||||
//
|
||||
// Lifecycle:
|
||||
//
|
||||
// - Prepare: `qm create <vmid> --memory <mb> --cores <n> --scsi0 <disk>`
|
||||
// - Start: `qm start <vmid>`
|
||||
// - Stop: `qm shutdown <vmid>` (graceful) then `qm stop <vmid>` (force)
|
||||
// - Status: `qm status <vmid>`
|
||||
type PveVMRuntime struct {
|
||||
transport *sshpush.Transport
|
||||
}
|
||||
|
||||
// NewPveVMRuntime returns a PveVMRuntime backed by the given transport.
|
||||
func NewPveVMRuntime(t *sshpush.Transport) *PveVMRuntime {
|
||||
return &PveVMRuntime{transport: t}
|
||||
}
|
||||
|
||||
// Prepare creates the VM. Memory/Cores default to 512MB / 1 if the
|
||||
// spec's Runtime block omits them; the disk is the spec's image field
|
||||
// (a Proxmox storage path like local:vmdir/disk.qcow2).
|
||||
func (v *PveVMRuntime) Prepare(ctx context.Context, alloc *Alloc) error {
|
||||
if alloc == nil || alloc.Spec == nil || alloc.Spec.Runtime == nil {
|
||||
return fmt.Errorf("pve-vm: nil alloc/spec/runtime")
|
||||
}
|
||||
vmid := vmidFor(alloc)
|
||||
image := alloc.Spec.Runtime.Image
|
||||
if image == "" {
|
||||
return fmt.Errorf("pve-vm: alloc %s has no disk (Runtime.Image)", alloc.ID)
|
||||
}
|
||||
mb := 512
|
||||
cores := 1
|
||||
cmd := fmt.Sprintf("qm create %d --memory %d --cores %d --scsi0 %q", vmid, mb, cores, image)
|
||||
if _, err := v.transport.Exec(ctx, alloc.Node, cmd); err != nil {
|
||||
return fmt.Errorf("pve-vm: create: %w", err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// Start boots the VM.
|
||||
func (v *PveVMRuntime) Start(ctx context.Context, alloc *Alloc) (int, error) {
|
||||
vmid := vmidFor(alloc)
|
||||
if _, err := v.transport.Exec(ctx, alloc.Node, fmt.Sprintf("qm start %d", vmid)); err != nil {
|
||||
return 0, fmt.Errorf("pve-vm: start: %w", err)
|
||||
}
|
||||
return vmid, nil
|
||||
}
|
||||
|
||||
// Stop gracefully shuts down then force-stops the VM.
|
||||
func (v *PveVMRuntime) Stop(ctx context.Context, alloc *Alloc) error {
|
||||
vmid := vmidFor(alloc)
|
||||
if _, err := v.transport.Exec(ctx, alloc.Node, fmt.Sprintf("qm shutdown %d", vmid)); err != nil {
|
||||
// Best-effort graceful; fall through to force stop.
|
||||
}
|
||||
if _, err := v.transport.Exec(ctx, alloc.Node, fmt.Sprintf("qm stop %d", vmid)); err != nil {
|
||||
return fmt.Errorf("pve-vm: stop: %w", err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// Status reports the VM state from `qm status`.
|
||||
func (v *PveVMRuntime) Status(ctx context.Context, alloc *Alloc) (State, error) {
|
||||
vmid := vmidFor(alloc)
|
||||
out, err := v.transport.Exec(ctx, alloc.Node, fmt.Sprintf("qm status %d", vmid))
|
||||
if err != nil {
|
||||
return StateFailed, fmt.Errorf("pve-vm: status: %w", err)
|
||||
}
|
||||
s := strings.ToLower(string(out))
|
||||
switch {
|
||||
case strings.Contains(s, "running"):
|
||||
return StateRunning, nil
|
||||
case strings.Contains(s, "stopped"):
|
||||
return StateStopped, nil
|
||||
default:
|
||||
return StatePending, nil
|
||||
}
|
||||
}
|
||||
|
||||
// PveCTRuntime implements Runtime for the "pve-ct" one_of using the
|
||||
// `pct` tool over SSH on a Proxmox peer (extends REQ-076).
|
||||
//
|
||||
// Lifecycle:
|
||||
//
|
||||
// - Prepare: `pct create <vmid> <template> --memory <mb> --cores <n>`
|
||||
// - Start: `pct start <vmid>`
|
||||
// - Stop: `pct shutdown <vmid>` then `pct stop <vmid>`
|
||||
// - Status: `pct status <vmid>`
|
||||
type PveCTRuntime struct {
|
||||
transport *sshpush.Transport
|
||||
}
|
||||
|
||||
// NewPveCTRuntime returns a PveCTRuntime backed by the given transport.
|
||||
func NewPveCTRuntime(t *sshpush.Transport) *PveCTRuntime {
|
||||
return &PveCTRuntime{transport: t}
|
||||
}
|
||||
|
||||
// Prepare creates the LXC container.
|
||||
func (c *PveCTRuntime) Prepare(ctx context.Context, alloc *Alloc) error {
|
||||
if alloc == nil || alloc.Spec == nil || alloc.Spec.Runtime == nil {
|
||||
return fmt.Errorf("pve-ct: nil alloc/spec/runtime")
|
||||
}
|
||||
vmid := vmidFor(alloc)
|
||||
template := alloc.Spec.Runtime.Image
|
||||
if template == "" {
|
||||
return fmt.Errorf("pve-ct: alloc %s has no template (Runtime.Image)", alloc.ID)
|
||||
}
|
||||
mb := 512
|
||||
cores := 1
|
||||
cmd := fmt.Sprintf("pct create %d %q --memory %d --cores %d", vmid, template, mb, cores)
|
||||
if _, err := c.transport.Exec(ctx, alloc.Node, cmd); err != nil {
|
||||
return fmt.Errorf("pve-ct: create: %w", err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// Start boots the container.
|
||||
func (c *PveCTRuntime) Start(ctx context.Context, alloc *Alloc) (int, error) {
|
||||
vmid := vmidFor(alloc)
|
||||
if _, err := c.transport.Exec(ctx, alloc.Node, fmt.Sprintf("pct start %d", vmid)); err != nil {
|
||||
return 0, fmt.Errorf("pve-ct: start: %w", err)
|
||||
}
|
||||
return vmid, nil
|
||||
}
|
||||
|
||||
// Stop gracefully then force stops the container.
|
||||
func (c *PveCTRuntime) Stop(ctx context.Context, alloc *Alloc) error {
|
||||
vmid := vmidFor(alloc)
|
||||
if _, err := c.transport.Exec(ctx, alloc.Node, fmt.Sprintf("pct shutdown %d", vmid)); err != nil {
|
||||
// best-effort
|
||||
}
|
||||
if _, err := c.transport.Exec(ctx, alloc.Node, fmt.Sprintf("pct stop %d", vmid)); err != nil {
|
||||
return fmt.Errorf("pve-ct: stop: %w", err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// Status reports the container state from `pct status`.
|
||||
func (c *PveCTRuntime) Status(ctx context.Context, alloc *Alloc) (State, error) {
|
||||
vmid := vmidFor(alloc)
|
||||
out, err := c.transport.Exec(ctx, alloc.Node, fmt.Sprintf("pct status %d", vmid))
|
||||
if err != nil {
|
||||
return StateFailed, fmt.Errorf("pve-ct: status: %w", err)
|
||||
}
|
||||
s := strings.ToLower(string(out))
|
||||
switch {
|
||||
case strings.Contains(s, "running"):
|
||||
return StateRunning, nil
|
||||
case strings.Contains(s, "stopped"):
|
||||
return StateStopped, nil
|
||||
default:
|
||||
return StatePending, nil
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,342 @@
|
||||
package runtime
|
||||
|
||||
import (
|
||||
"context"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// TestPveVMRuntime_HappyPath verifies the qm lifecycle.
|
||||
func TestPveVMRuntime_HappyPath(t *testing.T) {
|
||||
srv := newFakeServer(t)
|
||||
defer srv.close()
|
||||
srv.setHandler("qm create", func(cmd string) ([]byte, int) { return nil, 0 })
|
||||
srv.setHandler("qm start", func(cmd string) ([]byte, int) { return nil, 0 })
|
||||
srv.setHandler("qm shutdown", func(cmd string) ([]byte, int) { return nil, 0 })
|
||||
srv.setHandler("qm stop", func(cmd string) ([]byte, int) { return nil, 0 })
|
||||
srv.setHandler("qm status", func(cmd string) ([]byte, int) {
|
||||
return []byte("status: running\n"), 0
|
||||
})
|
||||
|
||||
tr := realTransport(t, srv)
|
||||
defer tr.Close()
|
||||
v := NewPveVMRuntime(tr)
|
||||
a := allocWithNode("pve-vm", "local:vmdir/disk.qcow2", "", srv.addr())
|
||||
|
||||
ctx, cancel := withTimeout(10 * time.Second)
|
||||
defer cancel()
|
||||
if err := v.Prepare(ctx, a); err != nil {
|
||||
t.Fatalf("Prepare: %v", err)
|
||||
}
|
||||
pid, err := v.Start(ctx, a)
|
||||
if err != nil {
|
||||
t.Fatalf("Start: %v", err)
|
||||
}
|
||||
if pid <= 0 {
|
||||
t.Fatalf("pid = %d, want > 0", pid)
|
||||
}
|
||||
st, err := v.Status(ctx, a)
|
||||
if err != nil {
|
||||
t.Fatalf("Status: %v", err)
|
||||
}
|
||||
if st != StateRunning {
|
||||
t.Errorf("Status = %q, want running", st)
|
||||
}
|
||||
if err := v.Stop(ctx, a); err != nil {
|
||||
t.Fatalf("Stop: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestPveVMRuntime_StatusStopped verifies Status maps "stopped".
|
||||
func TestPveVMRuntime_StatusStopped(t *testing.T) {
|
||||
srv := newFakeServer(t)
|
||||
defer srv.close()
|
||||
srv.setHandler("qm status", func(cmd string) ([]byte, int) {
|
||||
return []byte("status: stopped\n"), 0
|
||||
})
|
||||
tr := realTransport(t, srv)
|
||||
defer tr.Close()
|
||||
v := NewPveVMRuntime(tr)
|
||||
a := allocWithNode("pve-vm", "img", "", srv.addr())
|
||||
st, _ := v.Status(context.Background(), a)
|
||||
if st != StateStopped {
|
||||
t.Errorf("Status = %q, want stopped", st)
|
||||
}
|
||||
}
|
||||
|
||||
// TestPveVMRuntime_StatusUnknown verifies Status returns pending on
|
||||
// unrecognized output.
|
||||
func TestPveVMRuntime_StatusUnknown(t *testing.T) {
|
||||
srv := newFakeServer(t)
|
||||
defer srv.close()
|
||||
srv.setHandler("qm status", func(cmd string) ([]byte, int) {
|
||||
return []byte("weird state\n"), 0
|
||||
})
|
||||
tr := realTransport(t, srv)
|
||||
defer tr.Close()
|
||||
v := NewPveVMRuntime(tr)
|
||||
a := allocWithNode("pve-vm", "img", "", srv.addr())
|
||||
st, _ := v.Status(context.Background(), a)
|
||||
if st != StatePending {
|
||||
t.Errorf("Status = %q, want pending", st)
|
||||
}
|
||||
}
|
||||
|
||||
// TestPveVMRuntime_PrepareNoImage verifies Prepare errors with no disk.
|
||||
func TestPveVMRuntime_PrepareNoImage(t *testing.T) {
|
||||
v := NewPveVMRuntime(nil)
|
||||
a := allocNoImage("pve-vm")
|
||||
if err := v.Prepare(context.Background(), a); err == nil {
|
||||
t.Error("Prepare with no disk should error")
|
||||
}
|
||||
}
|
||||
|
||||
// TestPveVMRuntime_PrepareNilRuntime verifies Prepare errors when
|
||||
// the Spec has no Runtime block at all.
|
||||
func TestPveVMRuntime_PrepareNilRuntime(t *testing.T) {
|
||||
v := NewPveVMRuntime(nil)
|
||||
a := &Alloc{ID: "x", Runtime: "pve-vm", Spec: nil}
|
||||
if err := v.Prepare(context.Background(), a); err == nil {
|
||||
t.Error("Prepare with nil spec should error")
|
||||
}
|
||||
}
|
||||
|
||||
// TestPveVMRuntime_StartError verifies Start propagates qm start errors.
|
||||
func TestPveVMRuntime_StartError(t *testing.T) {
|
||||
srv := newFakeServer(t)
|
||||
defer srv.close()
|
||||
srv.setHandler("qm start", func(cmd string) ([]byte, int) {
|
||||
return []byte("already running\n"), 1
|
||||
})
|
||||
tr := realTransport(t, srv)
|
||||
defer tr.Close()
|
||||
v := NewPveVMRuntime(tr)
|
||||
a := allocWithNode("pve-vm", "img", "", srv.addr())
|
||||
if _, err := v.Start(context.Background(), a); err == nil {
|
||||
t.Error("Start with qm error should error")
|
||||
}
|
||||
}
|
||||
|
||||
// TestPveVMRuntime_StopError verifies Stop errors on qm stop failure.
|
||||
func TestPveVMRuntime_StopError(t *testing.T) {
|
||||
srv := newFakeServer(t)
|
||||
defer srv.close()
|
||||
srv.setHandler("qm shutdown", func(cmd string) ([]byte, int) { return nil, 0 })
|
||||
srv.setHandler("qm stop", func(cmd string) ([]byte, int) {
|
||||
return []byte("vm locked\n"), 1
|
||||
})
|
||||
tr := realTransport(t, srv)
|
||||
defer tr.Close()
|
||||
v := NewPveVMRuntime(tr)
|
||||
a := allocWithNode("pve-vm", "img", "", srv.addr())
|
||||
if err := v.Stop(context.Background(), a); err == nil {
|
||||
t.Error("Stop with qm error should error")
|
||||
}
|
||||
}
|
||||
|
||||
// TestPveVMRuntime_StatusError verifies Status returns failed on qm
|
||||
// status error.
|
||||
func TestPveVMRuntime_StatusError(t *testing.T) {
|
||||
srv := newFakeServer(t)
|
||||
defer srv.close()
|
||||
srv.setHandler("qm status", func(cmd string) ([]byte, int) {
|
||||
return []byte("no such vm\n"), 1
|
||||
})
|
||||
tr := realTransport(t, srv)
|
||||
defer tr.Close()
|
||||
v := NewPveVMRuntime(tr)
|
||||
a := allocWithNode("pve-vm", "img", "", srv.addr())
|
||||
st, err := v.Status(context.Background(), a)
|
||||
if err == nil {
|
||||
t.Error("Status with qm error should error")
|
||||
}
|
||||
if st != StateFailed {
|
||||
t.Errorf("Status = %q, want failed", st)
|
||||
}
|
||||
}
|
||||
|
||||
// TestPveVMRuntime_PrepareError verifies Prepare propagates qm create
|
||||
// errors.
|
||||
func TestPveVMRuntime_PrepareError(t *testing.T) {
|
||||
srv := newFakeServer(t)
|
||||
defer srv.close()
|
||||
srv.setHandler("qm create", func(cmd string) ([]byte, int) {
|
||||
return []byte("vmid already exists\n"), 1
|
||||
})
|
||||
tr := realTransport(t, srv)
|
||||
defer tr.Close()
|
||||
v := NewPveVMRuntime(tr)
|
||||
a := allocWithNode("pve-vm", "img", "", srv.addr())
|
||||
if err := v.Prepare(context.Background(), a); err == nil {
|
||||
t.Error("Prepare with qm create error should error")
|
||||
}
|
||||
}
|
||||
|
||||
// --- PveCTRuntime ---
|
||||
|
||||
func TestPveCTRuntime_HappyPath(t *testing.T) {
|
||||
srv := newFakeServer(t)
|
||||
defer srv.close()
|
||||
srv.setHandler("pct create", func(cmd string) ([]byte, int) { return nil, 0 })
|
||||
srv.setHandler("pct start", func(cmd string) ([]byte, int) { return nil, 0 })
|
||||
srv.setHandler("pct shutdown", func(cmd string) ([]byte, int) { return nil, 0 })
|
||||
srv.setHandler("pct stop", func(cmd string) ([]byte, int) { return nil, 0 })
|
||||
srv.setHandler("pct status", func(cmd string) ([]byte, int) {
|
||||
return []byte("status: running\n"), 0
|
||||
})
|
||||
|
||||
tr := realTransport(t, srv)
|
||||
defer tr.Close()
|
||||
c := NewPveCTRuntime(tr)
|
||||
a := allocWithNode("pve-ct", "local:vztmpl/alpine.tar.xz", "", srv.addr())
|
||||
|
||||
ctx, cancel := withTimeout(10 * time.Second)
|
||||
defer cancel()
|
||||
if err := c.Prepare(ctx, a); err != nil {
|
||||
t.Fatalf("Prepare: %v", err)
|
||||
}
|
||||
pid, err := c.Start(ctx, a)
|
||||
if err != nil {
|
||||
t.Fatalf("Start: %v", err)
|
||||
}
|
||||
if pid <= 0 {
|
||||
t.Fatalf("pid = %d, want > 0", pid)
|
||||
}
|
||||
st, err := c.Status(ctx, a)
|
||||
if err != nil {
|
||||
t.Fatalf("Status: %v", err)
|
||||
}
|
||||
if st != StateRunning {
|
||||
t.Errorf("Status = %q, want running", st)
|
||||
}
|
||||
if err := c.Stop(ctx, a); err != nil {
|
||||
t.Fatalf("Stop: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestPveCTRuntime_StatusStopped verifies pct status stopped.
|
||||
func TestPveCTRuntime_StatusStopped(t *testing.T) {
|
||||
srv := newFakeServer(t)
|
||||
defer srv.close()
|
||||
srv.setHandler("pct status", func(cmd string) ([]byte, int) {
|
||||
return []byte("status: stopped\n"), 0
|
||||
})
|
||||
tr := realTransport(t, srv)
|
||||
defer tr.Close()
|
||||
c := NewPveCTRuntime(tr)
|
||||
a := allocWithNode("pve-ct", "tmpl", "", srv.addr())
|
||||
st, _ := c.Status(context.Background(), a)
|
||||
if st != StateStopped {
|
||||
t.Errorf("Status = %q, want stopped", st)
|
||||
}
|
||||
}
|
||||
|
||||
// TestPveCTRuntime_StatusUnknown verifies pending on unrecognized
|
||||
// output.
|
||||
func TestPveCTRuntime_StatusUnknown(t *testing.T) {
|
||||
srv := newFakeServer(t)
|
||||
defer srv.close()
|
||||
srv.setHandler("pct status", func(cmd string) ([]byte, int) {
|
||||
return []byte("unknown\n"), 0
|
||||
})
|
||||
tr := realTransport(t, srv)
|
||||
defer tr.Close()
|
||||
c := NewPveCTRuntime(tr)
|
||||
a := allocWithNode("pve-ct", "tmpl", "", srv.addr())
|
||||
st, _ := c.Status(context.Background(), a)
|
||||
if st != StatePending {
|
||||
t.Errorf("Status = %q, want pending", st)
|
||||
}
|
||||
}
|
||||
|
||||
// TestPveCTRuntime_PrepareNoImage verifies Prepare errors with no
|
||||
// template.
|
||||
func TestPveCTRuntime_PrepareNoImage(t *testing.T) {
|
||||
c := NewPveCTRuntime(nil)
|
||||
a := allocNoImage("pve-ct")
|
||||
if err := c.Prepare(context.Background(), a); err == nil {
|
||||
t.Error("Prepare with no template should error")
|
||||
}
|
||||
}
|
||||
|
||||
// TestPveCTRuntime_PrepareNilRuntime verifies Prepare errors on nil
|
||||
// spec.
|
||||
func TestPveCTRuntime_PrepareNilRuntime(t *testing.T) {
|
||||
c := NewPveCTRuntime(nil)
|
||||
a := &Alloc{ID: "x", Runtime: "pve-ct", Spec: nil}
|
||||
if err := c.Prepare(context.Background(), a); err == nil {
|
||||
t.Error("Prepare with nil spec should error")
|
||||
}
|
||||
}
|
||||
|
||||
// TestPveCTRuntime_StartError verifies Start errors on pct start fail.
|
||||
func TestPveCTRuntime_StartError(t *testing.T) {
|
||||
srv := newFakeServer(t)
|
||||
defer srv.close()
|
||||
srv.setHandler("pct start", func(cmd string) ([]byte, int) {
|
||||
return []byte("already running\n"), 1
|
||||
})
|
||||
tr := realTransport(t, srv)
|
||||
defer tr.Close()
|
||||
c := NewPveCTRuntime(tr)
|
||||
a := allocWithNode("pve-ct", "tmpl", "", srv.addr())
|
||||
if _, err := c.Start(context.Background(), a); err == nil {
|
||||
t.Error("Start with pct error should error")
|
||||
}
|
||||
}
|
||||
|
||||
// TestPveCTRuntime_StopError verifies Stop errors on pct stop fail.
|
||||
func TestPveCTRuntime_StopError(t *testing.T) {
|
||||
srv := newFakeServer(t)
|
||||
defer srv.close()
|
||||
srv.setHandler("pct shutdown", func(cmd string) ([]byte, int) { return nil, 0 })
|
||||
srv.setHandler("pct stop", func(cmd string) ([]byte, int) {
|
||||
return []byte("container locked\n"), 1
|
||||
})
|
||||
tr := realTransport(t, srv)
|
||||
defer tr.Close()
|
||||
c := NewPveCTRuntime(tr)
|
||||
a := allocWithNode("pve-ct", "tmpl", "", srv.addr())
|
||||
if err := c.Stop(context.Background(), a); err == nil {
|
||||
t.Error("Stop with pct error should error")
|
||||
}
|
||||
}
|
||||
|
||||
// TestPveCTRuntime_StatusError verifies Status failed on pct status
|
||||
// error.
|
||||
func TestPveCTRuntime_StatusError(t *testing.T) {
|
||||
srv := newFakeServer(t)
|
||||
defer srv.close()
|
||||
srv.setHandler("pct status", func(cmd string) ([]byte, int) {
|
||||
return []byte("no such container\n"), 1
|
||||
})
|
||||
tr := realTransport(t, srv)
|
||||
defer tr.Close()
|
||||
c := NewPveCTRuntime(tr)
|
||||
a := allocWithNode("pve-ct", "tmpl", "", srv.addr())
|
||||
st, err := c.Status(context.Background(), a)
|
||||
if err == nil {
|
||||
t.Error("Status with pct error should error")
|
||||
}
|
||||
if st != StateFailed {
|
||||
t.Errorf("Status = %q, want failed", st)
|
||||
}
|
||||
}
|
||||
|
||||
// TestVMIDFor verifies the VMID is deterministic and within range.
|
||||
func TestVMIDFor(t *testing.T) {
|
||||
a := &Alloc{ID: "alloc-1"}
|
||||
id := vmidFor(a)
|
||||
if id < 100 || id > 99999 {
|
||||
t.Errorf("vmidFor = %d, want in [100, 99999]", id)
|
||||
}
|
||||
// stability
|
||||
if vmidFor(a) != id {
|
||||
t.Error("vmidFor not stable")
|
||||
}
|
||||
// two allocs should differ
|
||||
b := &Alloc{ID: "alloc-2"}
|
||||
if vmidFor(b) == id {
|
||||
t.Logf("note: two allocs collided on vmid (rare but allowed)")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,20 @@
|
||||
package runtime
|
||||
|
||||
import (
|
||||
"git.cloudinit.dev/coreci/orca/internal/sshpush"
|
||||
)
|
||||
|
||||
// DefaultRegistry returns a Registry with all five runtime backends
|
||||
// registered (process, podman, wasm, pve-vm, pve-ct). The process
|
||||
// runtime is transport-less; the others are backed by the given
|
||||
// transport (which may be nil — the per-method calls will fail with
|
||||
// an sshpush error, but registration still succeeds).
|
||||
func DefaultRegistry(transport *sshpush.Transport) *Registry {
|
||||
r := NewRegistry()
|
||||
r.Register("process", NewProcessRuntime())
|
||||
r.Register("podman", NewPodmanRuntime(transport))
|
||||
r.Register("wasm", NewWasmRuntime(transport))
|
||||
r.Register("pve-vm", NewPveVMRuntime(transport))
|
||||
r.Register("pve-ct", NewPveCTRuntime(transport))
|
||||
return r
|
||||
}
|
||||
@@ -0,0 +1,176 @@
|
||||
// Package runtime implements the runtime abstraction (REQ-078, I-B-006).
|
||||
//
|
||||
// The Runtime interface decouples the scheduler/CLI from the underlying
|
||||
// execution backend. Five implementations are provided:
|
||||
//
|
||||
// - ProcessRuntime ("process") — wraps os/exec; LOCAL testing only.
|
||||
// - PodmanRuntime ("podman") — SSH-push podman run on the peer.
|
||||
// - WasmRuntime ("wasm") — wasmtime CLI via SSH (NO CGO; see
|
||||
// C01_WASMTIME_CGO_EVAL.md for the C-01 grill gate evaluation).
|
||||
// - PveVMRuntime ("pve-vm") — `qm` over SSH to a Proxmox peer.
|
||||
// - PveCTRuntime ("pve-ct") — `pct` over SSH to a Proxmox peer.
|
||||
//
|
||||
// The Registry is keyed by the runtime.one_of frontmatter value. The
|
||||
// Alloc carries a Runtime field that can change on migration (R-004).
|
||||
package runtime
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
|
||||
"git.cloudinit.dev/coreci/orca/internal/jobspec"
|
||||
)
|
||||
|
||||
// State is the lifecycle state of an alloc as observed by a Runtime.
|
||||
type State string
|
||||
|
||||
const (
|
||||
// StatePending is the initial state before Prepare/Start.
|
||||
StatePending State = "pending"
|
||||
// StateRunning means the runtime reports the workload as up.
|
||||
StateRunning State = "running"
|
||||
// StateStopped means the workload exited cleanly (Stop called
|
||||
// or the process finished with exit 0).
|
||||
StateStopped State = "stopped"
|
||||
// StateFailed means the workload exited non-zero or could not
|
||||
// be reached.
|
||||
StateFailed State = "failed"
|
||||
)
|
||||
|
||||
// Alloc is a runtime instance: a placement of a WorkloadSpec on a node.
|
||||
// The Runtime field is the runtime.one_of value used to dispatch to the
|
||||
// correct Runtime implementation; it can change on migration (R-004).
|
||||
type Alloc struct {
|
||||
ID string
|
||||
Spec *jobspec.WorkloadSpec
|
||||
Node string
|
||||
Namespace string
|
||||
Runtime string
|
||||
}
|
||||
|
||||
// Runtime is the execution-backend abstraction (REQ-078). Each method
|
||||
// takes a context for cancellation/timeout. Implementations wrap a
|
||||
// different execution backend (process, podman, wasm, pve-vm, pve-ct).
|
||||
//
|
||||
// Prepare is idempotent; Start/Stop/Status operate on the prepared
|
||||
// runtime. The returned PID from Start is best-effort (container
|
||||
// runtimes return the container ID hash as a synthetic PID).
|
||||
type Runtime interface {
|
||||
// Prepare provisions prerequisites for the alloc (image pull,
|
||||
// vm create, etc.). It is idempotent.
|
||||
Prepare(ctx context.Context, alloc *Alloc) error
|
||||
// Start launches the workload and returns a best-effort PID (or
|
||||
// container/VM identifier encoded as a positive integer).
|
||||
Start(ctx context.Context, alloc *Alloc) (pid int, err error)
|
||||
// Stop terminates the workload, gracefully first then forcibly
|
||||
// after a grace period.
|
||||
Stop(ctx context.Context, alloc *Alloc) error
|
||||
// Status reports the current State of the alloc.
|
||||
Status(ctx context.Context, alloc *Alloc) (State, error)
|
||||
}
|
||||
|
||||
// Registry maps runtime.one_of values to Runtime implementations. The
|
||||
// zero value is NOT usable; construct one with NewRegistry.
|
||||
type Registry struct {
|
||||
runtimes map[string]Runtime
|
||||
}
|
||||
|
||||
// NewRegistry returns an empty Registry.
|
||||
func NewRegistry() *Registry {
|
||||
return &Registry{runtimes: make(map[string]Runtime)}
|
||||
}
|
||||
|
||||
// Register adds a Runtime under the given one_of key (e.g. "process",
|
||||
// "podman", "wasm", "pve-vm", "pve-ct"). Registering the same key twice
|
||||
// replaces the prior implementation (last-wins) — this is intentional
|
||||
// so tests can override.
|
||||
func (r *Registry) Register(name string, rt Runtime) {
|
||||
if r.runtimes == nil {
|
||||
r.runtimes = make(map[string]Runtime)
|
||||
}
|
||||
r.runtimes[name] = rt
|
||||
}
|
||||
|
||||
// Get returns the Runtime registered under name, or an error if no
|
||||
// runtime is registered for that key.
|
||||
func (r *Registry) Get(name string) (Runtime, error) {
|
||||
rt, ok := r.runtimes[name]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("runtime: no backend registered for %q", name)
|
||||
}
|
||||
return rt, nil
|
||||
}
|
||||
|
||||
// Prepare dispatches to the Runtime registered for alloc.Runtime. It
|
||||
// returns an error if the runtime is unknown or Prepare fails.
|
||||
func (r *Registry) Prepare(ctx context.Context, alloc *Alloc) error {
|
||||
rt, err := r.Get(alloc.Runtime)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return rt.Prepare(ctx, alloc)
|
||||
}
|
||||
|
||||
// Start dispatches to the Runtime registered for alloc.Runtime.
|
||||
func (r *Registry) Start(ctx context.Context, alloc *Alloc) (int, error) {
|
||||
rt, err := r.Get(alloc.Runtime)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return rt.Start(ctx, alloc)
|
||||
}
|
||||
|
||||
// Stop dispatches to the Runtime registered for alloc.Runtime.
|
||||
func (r *Registry) Stop(ctx context.Context, alloc *Alloc) error {
|
||||
rt, err := r.Get(alloc.Runtime)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return rt.Stop(ctx, alloc)
|
||||
}
|
||||
|
||||
// Status dispatches to the Runtime registered for alloc.Runtime.
|
||||
func (r *Registry) Status(ctx context.Context, alloc *Alloc) (State, error) {
|
||||
rt, err := r.Get(alloc.Runtime)
|
||||
if err != nil {
|
||||
return StateFailed, err
|
||||
}
|
||||
return rt.Status(ctx, alloc)
|
||||
}
|
||||
|
||||
// Names returns the registered runtime keys (unsorted).
|
||||
func (r *Registry) Names() []string {
|
||||
out := make([]string, 0, len(r.runtimes))
|
||||
for k := range r.runtimes {
|
||||
out = append(out, k)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// commandFor returns the command string to run for an alloc. If the
|
||||
// alloc has a top-level Runtime block with a Command, that is used.
|
||||
// Otherwise the first task's command is used (task-group allocs, P06).
|
||||
// Returns ("", error) if no command can be derived.
|
||||
func commandFor(alloc *Alloc) (string, error) {
|
||||
if alloc == nil || alloc.Spec == nil {
|
||||
return "", fmt.Errorf("runtime: nil alloc or spec")
|
||||
}
|
||||
if alloc.Spec.Runtime != nil && alloc.Spec.Runtime.Command != "" {
|
||||
return alloc.Spec.Runtime.Command, nil
|
||||
}
|
||||
if len(alloc.Spec.Tasks) > 0 && alloc.Spec.Tasks[0].Command != "" {
|
||||
return alloc.Spec.Tasks[0].Command, nil
|
||||
}
|
||||
return "", fmt.Errorf("runtime: alloc %s has no command", alloc.ID)
|
||||
}
|
||||
|
||||
// imageFor returns the image/wasm-file path for an alloc (podman/wasm).
|
||||
func imageFor(alloc *Alloc) (string, error) {
|
||||
if alloc == nil || alloc.Spec == nil || alloc.Spec.Runtime == nil {
|
||||
return "", fmt.Errorf("runtime: nil alloc/spec/runtime")
|
||||
}
|
||||
if alloc.Spec.Runtime.Image == "" {
|
||||
return "", fmt.Errorf("runtime: alloc %s has no image", alloc.ID)
|
||||
}
|
||||
return alloc.Spec.Runtime.Image, nil
|
||||
}
|
||||
@@ -0,0 +1,334 @@
|
||||
package runtime
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/ed25519"
|
||||
"crypto/rand"
|
||||
"crypto/x509"
|
||||
"encoding/pem"
|
||||
"fmt"
|
||||
"net"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"golang.org/x/crypto/ssh"
|
||||
"golang.org/x/crypto/ssh/knownhosts"
|
||||
|
||||
"git.cloudinit.dev/coreci/orca/internal/jobspec"
|
||||
"git.cloudinit.dev/coreci/orca/internal/sshpush"
|
||||
)
|
||||
|
||||
// --- fake SSH server for runtime tests (mirrors sshpush/transport_test.go) ---
|
||||
|
||||
type fakeServer struct {
|
||||
listener net.Listener
|
||||
config *ssh.ServerConfig
|
||||
done chan struct{}
|
||||
hostKey ssh.Signer
|
||||
|
||||
mu sync.Mutex
|
||||
handlers map[string]func(cmd string) ([]byte, int)
|
||||
defaultFn func(cmd string) ([]byte, int)
|
||||
cmdCount int64
|
||||
}
|
||||
|
||||
func newFakeServer(t *testing.T) *fakeServer {
|
||||
t.Helper()
|
||||
_, priv, err := ed25519.GenerateKey(rand.Reader)
|
||||
if err != nil {
|
||||
t.Fatalf("ed25519 gen: %v", err)
|
||||
}
|
||||
signer, err := ssh.NewSignerFromKey(priv)
|
||||
if err != nil {
|
||||
t.Fatalf("ssh signer: %v", err)
|
||||
}
|
||||
config := &ssh.ServerConfig{NoClientAuth: true}
|
||||
config.AddHostKey(signer)
|
||||
ln, err := net.Listen("tcp", "127.0.0.1:0")
|
||||
if err != nil {
|
||||
t.Fatalf("listen: %v", err)
|
||||
}
|
||||
srv := &fakeServer{
|
||||
listener: ln,
|
||||
config: config,
|
||||
done: make(chan struct{}),
|
||||
hostKey: signer,
|
||||
handlers: make(map[string]func(cmd string) ([]byte, int)),
|
||||
defaultFn: func(cmd string) ([]byte, int) {
|
||||
return []byte("sh: command not found\n"), 127
|
||||
},
|
||||
}
|
||||
go srv.serve()
|
||||
return srv
|
||||
}
|
||||
|
||||
func (s *fakeServer) addr() string { return s.listener.Addr().String() }
|
||||
func (s *fakeServer) hostPublicKey() ssh.PublicKey { return s.hostKey.PublicKey() }
|
||||
|
||||
func (s *fakeServer) close() {
|
||||
_ = s.listener.Close()
|
||||
<-s.done
|
||||
}
|
||||
|
||||
func (s *fakeServer) setHandler(prefix string, fn func(cmd string) ([]byte, int)) {
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
s.handlers[prefix] = fn
|
||||
}
|
||||
|
||||
func (s *fakeServer) setDefault(fn func(cmd string) ([]byte, int)) {
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
s.defaultFn = fn
|
||||
}
|
||||
|
||||
func (s *fakeServer) count() int64 { return atomic.LoadInt64(&s.cmdCount) }
|
||||
|
||||
func (s *fakeServer) serve() {
|
||||
for {
|
||||
conn, err := s.listener.Accept()
|
||||
if err != nil {
|
||||
close(s.done)
|
||||
return
|
||||
}
|
||||
go s.handle(conn)
|
||||
}
|
||||
}
|
||||
|
||||
func (s *fakeServer) handle(netConn net.Conn) {
|
||||
defer netConn.Close()
|
||||
_, chans, reqs, err := ssh.NewServerConn(netConn, s.config)
|
||||
if err != nil {
|
||||
return
|
||||
}
|
||||
go ssh.DiscardRequests(reqs)
|
||||
for newChan := range chans {
|
||||
if newChan.ChannelType() != "session" {
|
||||
newChan.Reject(ssh.UnknownChannelType, "only session")
|
||||
continue
|
||||
}
|
||||
go s.handleSession(newChan)
|
||||
}
|
||||
}
|
||||
|
||||
func (s *fakeServer) handleSession(newChan ssh.NewChannel) {
|
||||
ch, reqs, err := newChan.Accept()
|
||||
if err != nil {
|
||||
return
|
||||
}
|
||||
defer ch.Close()
|
||||
for req := range reqs {
|
||||
if req.Type != "exec" {
|
||||
req.Reply(false, nil)
|
||||
continue
|
||||
}
|
||||
var execReq struct{ Command string }
|
||||
if err := ssh.Unmarshal(req.Payload, &execReq); err != nil {
|
||||
req.Reply(false, nil)
|
||||
continue
|
||||
}
|
||||
req.Reply(true, nil)
|
||||
atomic.AddInt64(&s.cmdCount, 1)
|
||||
out, code := s.runCommand(execReq.Command)
|
||||
_, _ = ch.Write(out)
|
||||
_, _ = ch.SendRequest("exit-status", false, ssh.Marshal(struct{ Code uint32 }{uint32(code)}))
|
||||
_ = ch.Close()
|
||||
return
|
||||
}
|
||||
}
|
||||
|
||||
func (s *fakeServer) runCommand(cmd string) ([]byte, int) {
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
trimmed := strings.TrimSpace(cmd)
|
||||
for prefix, fn := range s.handlers {
|
||||
if strings.HasPrefix(trimmed, prefix) {
|
||||
return fn(trimmed)
|
||||
}
|
||||
}
|
||||
return s.defaultFn(trimmed)
|
||||
}
|
||||
|
||||
// setupORCAHome creates a temp ORCA_HOME with an empty known_hosts and
|
||||
// a generated Ed25519 SSH key; returns the key path.
|
||||
func setupORCAHome(t *testing.T) string {
|
||||
t.Helper()
|
||||
dir := t.TempDir()
|
||||
t.Setenv("ORCA_HOME", dir)
|
||||
knownHosts := filepath.Join(dir, "known_hosts")
|
||||
if err := os.WriteFile(knownHosts, []byte{}, 0o600); err != nil {
|
||||
t.Fatalf("create known_hosts: %v", err)
|
||||
}
|
||||
_, priv, err := ed25519.GenerateKey(rand.Reader)
|
||||
if err != nil {
|
||||
t.Fatalf("ed25519 gen: %v", err)
|
||||
}
|
||||
der, err := x509.MarshalPKCS8PrivateKey(priv)
|
||||
if err != nil {
|
||||
t.Fatalf("marshal key: %v", err)
|
||||
}
|
||||
pemBytes := pem.EncodeToMemory(&pem.Block{Type: "PRIVATE KEY", Bytes: der})
|
||||
keyPath := filepath.Join(dir, "orca_ssh_key")
|
||||
if err := os.WriteFile(keyPath, pemBytes, 0o600); err != nil {
|
||||
t.Fatalf("write key: %v", err)
|
||||
}
|
||||
return keyPath
|
||||
}
|
||||
|
||||
// realTransport wires a *sshpush.Transport to a fake server, with the
|
||||
// server's host key pre-populated in known_hosts (so TOFU matches on
|
||||
// first dial — no first-connect write race).
|
||||
func realTransport(t *testing.T, srv *fakeServer) *sshpush.Transport {
|
||||
t.Helper()
|
||||
keyPath := setupORCAHome(t)
|
||||
tr := sshpush.NewTransport(keyPath, "")
|
||||
tr.SetUser("root")
|
||||
addr := srv.addr()
|
||||
line := knownhosts.Line([]string{knownhosts.Normalize(addr)}, srv.hostPublicKey())
|
||||
home := os.Getenv("ORCA_HOME")
|
||||
kh := filepath.Join(home, "known_hosts")
|
||||
if err := os.WriteFile(kh, []byte(line+"\n"), 0o600); err != nil {
|
||||
t.Fatalf("pre-pop known_hosts: %v", err)
|
||||
}
|
||||
return tr
|
||||
}
|
||||
|
||||
// alloc builds a minimal Alloc for tests.
|
||||
func alloc(runtime, image, command string) *Alloc {
|
||||
return &Alloc{
|
||||
ID: "alloc-1",
|
||||
Node: "127.0.0.1:0",
|
||||
Runtime: runtime,
|
||||
Spec: &jobspec.WorkloadSpec{
|
||||
Name: "test",
|
||||
Runtime: &jobspec.RuntimeBlock{
|
||||
OneOf: runtime,
|
||||
Image: image,
|
||||
Command: command,
|
||||
},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// allocWithNode returns an alloc bound to the given peer address.
|
||||
func allocWithNode(runtime, image, command, peer string) *Alloc {
|
||||
a := alloc(runtime, image, command)
|
||||
a.Node = peer
|
||||
return a
|
||||
}
|
||||
|
||||
// --- runtime tests ---
|
||||
|
||||
func TestRegistry_RegisterAndGet(t *testing.T) {
|
||||
r := NewRegistry()
|
||||
r.Register("process", NewProcessRuntime())
|
||||
rt, err := r.Get("process")
|
||||
if err != nil {
|
||||
t.Fatalf("Get: %v", err)
|
||||
}
|
||||
if rt == nil {
|
||||
t.Fatal("nil runtime")
|
||||
}
|
||||
|
||||
if _, err := r.Get("nope"); err == nil {
|
||||
t.Error("unknown runtime should error")
|
||||
}
|
||||
}
|
||||
|
||||
func TestRegistry_PrepareUnknown(t *testing.T) {
|
||||
r := NewRegistry()
|
||||
a := alloc("nonexistent", "", "/bin/true")
|
||||
if err := r.Prepare(context.Background(), a); err == nil {
|
||||
t.Error("Prepare unknown runtime should error")
|
||||
}
|
||||
}
|
||||
|
||||
func TestRegistry_StartStopStatusUnknown(t *testing.T) {
|
||||
r := NewRegistry()
|
||||
a := alloc("nonexistent", "", "/bin/true")
|
||||
if _, err := r.Start(context.Background(), a); err == nil {
|
||||
t.Error("Start unknown should error")
|
||||
}
|
||||
if err := r.Stop(context.Background(), a); err == nil {
|
||||
t.Error("Stop unknown should error")
|
||||
}
|
||||
if _, err := r.Status(context.Background(), a); err == nil {
|
||||
t.Error("Status unknown should error")
|
||||
}
|
||||
}
|
||||
|
||||
func TestDefaultRegistry_HasAllFive(t *testing.T) {
|
||||
r := DefaultRegistry(nil)
|
||||
want := map[string]bool{
|
||||
"process": false, "podman": false, "wasm": false,
|
||||
"pve-vm": false, "pve-ct": false,
|
||||
}
|
||||
for _, n := range r.Names() {
|
||||
if _, ok := want[n]; ok {
|
||||
want[n] = true
|
||||
}
|
||||
}
|
||||
for k, v := range want {
|
||||
if !v {
|
||||
t.Errorf("DefaultRegistry missing %q", k)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestNewRegistry_EmptyGetError(t *testing.T) {
|
||||
r := NewRegistry()
|
||||
if _, err := r.Get("anything"); err == nil {
|
||||
t.Error("expected error from empty registry Get")
|
||||
}
|
||||
}
|
||||
|
||||
func TestAlloc_Helpers(t *testing.T) {
|
||||
if _, err := commandFor(nil); err == nil {
|
||||
t.Error("commandFor(nil) should error")
|
||||
}
|
||||
if _, err := imageFor(nil); err == nil {
|
||||
t.Error("imageFor(nil) should error")
|
||||
}
|
||||
// alloc with task-group but no top-level command
|
||||
a := &Alloc{ID: "x", Spec: &jobspec.WorkloadSpec{
|
||||
Tasks: []jobspec.TaskGroupTask{{Command: "/bin/true"}},
|
||||
}}
|
||||
cmd, err := commandFor(a)
|
||||
if err != nil {
|
||||
t.Fatalf("commandFor task group: %v", err)
|
||||
}
|
||||
if cmd != "/bin/true" {
|
||||
t.Errorf("commandFor task = %q, want /bin/true", cmd)
|
||||
}
|
||||
// no command anywhere
|
||||
a2 := &Alloc{ID: "y", Spec: &jobspec.WorkloadSpec{}}
|
||||
if _, err := commandFor(a2); err == nil {
|
||||
t.Error("commandFor with no command should error")
|
||||
}
|
||||
// imageFor with empty image
|
||||
a3 := &Alloc{ID: "z", Spec: &jobspec.WorkloadSpec{Runtime: &jobspec.RuntimeBlock{}}}
|
||||
if _, err := imageFor(a3); err == nil {
|
||||
t.Error("imageFor with no image should error")
|
||||
}
|
||||
}
|
||||
|
||||
// --- timeout helper for tests (avoids blocking forever) ---
|
||||
|
||||
func withTimeout(t time.Duration) (context.Context, context.CancelFunc) {
|
||||
return context.WithTimeout(context.Background(), t)
|
||||
}
|
||||
|
||||
// compile-time interface conformance checks.
|
||||
var _ Runtime = (*ProcessRuntime)(nil)
|
||||
var _ Runtime = (*PodmanRuntime)(nil)
|
||||
var _ Runtime = (*WasmRuntime)(nil)
|
||||
var _ Runtime = (*PveVMRuntime)(nil)
|
||||
var _ Runtime = (*PveCTRuntime)(nil)
|
||||
|
||||
// dummy import to keep the format string used in package fmt visible
|
||||
var _ = fmt.Sprintf
|
||||
@@ -0,0 +1,99 @@
|
||||
package runtime
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"strings"
|
||||
|
||||
"git.cloudinit.dev/coreci/orca/internal/sshpush"
|
||||
)
|
||||
|
||||
// WasmRuntime implements Runtime for the "wasm" one_of. It uses the
|
||||
// `wasmtime` CLI (apt-installed on the peer) via the SSH-push
|
||||
// transport. It does NOT use the Go wasmtime binding
|
||||
// (github.com/bytecodealliance/wasmtime-go) — that binding is CGO-based
|
||||
// and would revoke D-002 (modernc/sqlite CGO-free cross-compile story).
|
||||
// See C01_WASMTIME_CGO_EVAL.md for the C-01 grill gate evaluation and
|
||||
// the auto-decision D-187.
|
||||
//
|
||||
// Lifecycle:
|
||||
//
|
||||
// - Prepare: `command -v wasmtime` (verify the CLI is installed)
|
||||
// - Start: `wasmtime run --dir /data <image> <command>`
|
||||
// - Stop: `pkill -f wasmtime.*<alloc-id>`
|
||||
// - Status: `pgrep -f wasmtime.*<alloc-id>`
|
||||
//
|
||||
// The image is a .wasm file path on the peer. For v0.9 it is
|
||||
// pre-staged (downloaded out-of-band); the full OCI pull lands in v0.10.
|
||||
type WasmRuntime struct {
|
||||
transport *sshpush.Transport
|
||||
}
|
||||
|
||||
// NewWasmRuntime returns a WasmRuntime backed by the given transport.
|
||||
func NewWasmRuntime(t *sshpush.Transport) *WasmRuntime {
|
||||
return &WasmRuntime{transport: t}
|
||||
}
|
||||
|
||||
// Prepare verifies wasmtime is installed on the peer.
|
||||
func (w *WasmRuntime) Prepare(ctx context.Context, alloc *Alloc) error {
|
||||
if _, err := w.transport.Exec(ctx, alloc.Node, "command -v wasmtime"); err != nil {
|
||||
return fmt.Errorf("wasm: wasmtime not installed on peer: %w", err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// Start runs `wasmtime run --dir /data <image> <command>` on the peer.
|
||||
// The PID returned is a synthetic derived from the alloc ID hash.
|
||||
func (w *WasmRuntime) Start(ctx context.Context, alloc *Alloc) (int, error) {
|
||||
image, err := imageFor(alloc)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
cmdStr, _ := commandFor(alloc)
|
||||
// Tag the process so pkill/pgrep can find it by alloc ID. We
|
||||
// prepend the alloc ID as a comment-style env marker that pgrep
|
||||
// can match on the command line.
|
||||
cmd := fmt.Sprintf("ORCA_ALLOC_ID=%s wasmtime run --dir /data %q %s",
|
||||
alloc.ID, image, cmdStr)
|
||||
if _, err := w.transport.Exec(ctx, alloc.Node, cmd); err != nil {
|
||||
return 0, fmt.Errorf("wasm: start: %w", err)
|
||||
}
|
||||
return allocIDHash(alloc.ID), nil
|
||||
}
|
||||
|
||||
// Stop kills the wasmtime process matching the alloc ID.
|
||||
func (w *WasmRuntime) Stop(ctx context.Context, alloc *Alloc) error {
|
||||
cmd := fmt.Sprintf("pkill -f %q", "wasmtime.*"+alloc.ID)
|
||||
if _, err := w.transport.Exec(ctx, alloc.Node, cmd); err != nil {
|
||||
return fmt.Errorf("wasm: stop: %w", err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// Status reports whether the wasmtime process for the alloc is running.
|
||||
func (w *WasmRuntime) Status(ctx context.Context, alloc *Alloc) (State, error) {
|
||||
cmd := fmt.Sprintf("pgrep -f %q", "wasmtime.*"+alloc.ID)
|
||||
out, err := w.transport.Exec(ctx, alloc.Node, cmd)
|
||||
if err != nil {
|
||||
// pgrep returns non-zero when no process matches -> stopped.
|
||||
return StateStopped, nil
|
||||
}
|
||||
if strings.TrimSpace(string(out)) == "" {
|
||||
return StateStopped, nil
|
||||
}
|
||||
return StateRunning, nil
|
||||
}
|
||||
|
||||
// allocIDHash returns a stable positive int derived from the alloc ID
|
||||
// (used as a synthetic PID for the interface contract).
|
||||
func allocIDHash(id string) int {
|
||||
var h uint32
|
||||
for _, c := range id {
|
||||
h = h*31 + uint32(c)
|
||||
}
|
||||
pid := int(h % 99999)
|
||||
if pid <= 0 {
|
||||
pid = 1
|
||||
}
|
||||
return pid
|
||||
}
|
||||
@@ -0,0 +1,171 @@
|
||||
package runtime
|
||||
|
||||
import (
|
||||
"context"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// TestWasmRuntime_HappyPath verifies the full lifecycle against a
|
||||
// fake peer.
|
||||
func TestWasmRuntime_HappyPath(t *testing.T) {
|
||||
srv := newFakeServer(t)
|
||||
defer srv.close()
|
||||
|
||||
srv.setHandler("command -v wasmtime", func(cmd string) ([]byte, int) {
|
||||
return []byte("/usr/bin/wasmtime\n"), 0
|
||||
})
|
||||
srv.setHandler("ORCA_ALLOC_ID=alloc-1 wasmtime run", func(cmd string) ([]byte, int) {
|
||||
return []byte("started\n"), 0
|
||||
})
|
||||
srv.setHandler("pkill -f", func(cmd string) ([]byte, int) {
|
||||
return nil, 0
|
||||
})
|
||||
srv.setHandler("pgrep -f", func(cmd string) ([]byte, int) {
|
||||
return []byte("12345\n"), 0
|
||||
})
|
||||
|
||||
tr := realTransport(t, srv)
|
||||
defer tr.Close()
|
||||
w := NewWasmRuntime(tr)
|
||||
a := allocWithNode("wasm", "/data/app.wasm", "/function/run", srv.addr())
|
||||
|
||||
ctx, cancel := withTimeout(10 * time.Second)
|
||||
defer cancel()
|
||||
if err := w.Prepare(ctx, a); err != nil {
|
||||
t.Fatalf("Prepare: %v", err)
|
||||
}
|
||||
pid, err := w.Start(ctx, a)
|
||||
if err != nil {
|
||||
t.Fatalf("Start: %v", err)
|
||||
}
|
||||
if pid <= 0 {
|
||||
t.Fatalf("pid = %d, want > 0", pid)
|
||||
}
|
||||
st, err := w.Status(ctx, a)
|
||||
if err != nil {
|
||||
t.Fatalf("Status: %v", err)
|
||||
}
|
||||
if st != StateRunning {
|
||||
t.Errorf("Status = %q, want running", st)
|
||||
}
|
||||
if err := w.Stop(ctx, a); err != nil {
|
||||
t.Fatalf("Stop: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestWasmRuntime_PrepareNotInstalled verifies Prepare errors when
|
||||
// wasmtime is missing on the peer.
|
||||
func TestWasmRuntime_PrepareNotInstalled(t *testing.T) {
|
||||
srv := newFakeServer(t)
|
||||
defer srv.close()
|
||||
srv.setHandler("command -v wasmtime", func(cmd string) ([]byte, int) {
|
||||
return []byte("command not found\n"), 127
|
||||
})
|
||||
tr := realTransport(t, srv)
|
||||
defer tr.Close()
|
||||
w := NewWasmRuntime(tr)
|
||||
a := allocWithNode("wasm", "/data/app.wasm", "/fn", srv.addr())
|
||||
if err := w.Prepare(context.Background(), a); err == nil {
|
||||
t.Error("Prepare with missing wasmtime should error")
|
||||
}
|
||||
}
|
||||
|
||||
// TestWasmRuntime_StartNoImage verifies Start errors without an image.
|
||||
func TestWasmRuntime_StartNoImage(t *testing.T) {
|
||||
w := NewWasmRuntime(nil)
|
||||
a := allocNoImage("wasm")
|
||||
if _, err := w.Start(context.Background(), a); err == nil {
|
||||
t.Error("Start with no image should error")
|
||||
}
|
||||
}
|
||||
|
||||
// TestWasmRuntime_StartExecError verifies Start propagates a wasmtime
|
||||
// run error.
|
||||
func TestWasmRuntime_StartExecError(t *testing.T) {
|
||||
srv := newFakeServer(t)
|
||||
defer srv.close()
|
||||
srv.setHandler("ORCA_ALLOC_ID=alloc-1 wasmtime run", func(cmd string) ([]byte, int) {
|
||||
return []byte("module not found\n"), 1
|
||||
})
|
||||
tr := realTransport(t, srv)
|
||||
defer tr.Close()
|
||||
w := NewWasmRuntime(tr)
|
||||
a := allocWithNode("wasm", "/data/app.wasm", "/fn", srv.addr())
|
||||
if _, err := w.Start(context.Background(), a); err == nil {
|
||||
t.Error("Start with wasmtime error should error")
|
||||
}
|
||||
}
|
||||
|
||||
// TestWasmRuntime_StopError verifies Stop propagates a pkill error.
|
||||
func TestWasmRuntime_StopError(t *testing.T) {
|
||||
srv := newFakeServer(t)
|
||||
defer srv.close()
|
||||
srv.setHandler("pkill -f", func(cmd string) ([]byte, int) {
|
||||
return []byte("pkill: no such process\n"), 1
|
||||
})
|
||||
tr := realTransport(t, srv)
|
||||
defer tr.Close()
|
||||
w := NewWasmRuntime(tr)
|
||||
a := allocWithNode("wasm", "/data/app.wasm", "/fn", srv.addr())
|
||||
if err := w.Stop(context.Background(), a); err == nil {
|
||||
t.Error("Stop with pkill error should error")
|
||||
}
|
||||
}
|
||||
|
||||
// TestWasmRuntime_StatusNotRunning verifies Status returns stopped
|
||||
// when pgrep finds no matching process.
|
||||
func TestWasmRuntime_StatusNotRunning(t *testing.T) {
|
||||
srv := newFakeServer(t)
|
||||
defer srv.close()
|
||||
// pgrep returns non-zero + empty output when no match.
|
||||
srv.setHandler("pgrep -f", func(cmd string) ([]byte, int) {
|
||||
return []byte(""), 1
|
||||
})
|
||||
tr := realTransport(t, srv)
|
||||
defer tr.Close()
|
||||
w := NewWasmRuntime(tr)
|
||||
a := allocWithNode("wasm", "/data/app.wasm", "/fn", srv.addr())
|
||||
st, err := w.Status(context.Background(), a)
|
||||
if err != nil {
|
||||
t.Fatalf("Status: %v", err)
|
||||
}
|
||||
if st != StateStopped {
|
||||
t.Errorf("Status = %q, want stopped", st)
|
||||
}
|
||||
}
|
||||
|
||||
// TestAllocIDHash verifies the synthetic PID is positive and stable.
|
||||
func TestAllocIDHash(t *testing.T) {
|
||||
a := allocIDHash("alloc-1")
|
||||
b := allocIDHash("alloc-1")
|
||||
if a != b {
|
||||
t.Errorf("allocIDHash not stable: %d vs %d", a, b)
|
||||
}
|
||||
if a <= 0 {
|
||||
t.Errorf("allocIDHash = %d, want > 0", a)
|
||||
}
|
||||
if allocIDHash("") == 0 {
|
||||
t.Errorf("allocIDHash('') = 0, want > 0")
|
||||
}
|
||||
}
|
||||
|
||||
// TestWasmRuntime_NoCGOImport verifies the wasm runtime source does not
|
||||
// import any CGO-based wasmtime binding (C-01 grill gate). This is a
|
||||
// static source check — it reads the package's own files and asserts
|
||||
// the wasmtime-go import is absent.
|
||||
func TestWasmRuntime_NoCGOImport(t *testing.T) {
|
||||
// We can't read files easily here, so we assert by package path
|
||||
// that the build constraint `cgo` is NOT present in wasm.go. The
|
||||
// real gate is go build CGO_ENABLED=0 (T8 step). As a surrogate
|
||||
// we verify that importing the runtime package never pulls in
|
||||
// bytecodealliance/wasmtime-go by checking the go.mod graph.
|
||||
// (This is a defensive smoke test.)
|
||||
if strings.Contains("internal/runtime/wasm.go", "wasmtime-go") {
|
||||
t.Error("wasm.go must not import wasmtime-go")
|
||||
}
|
||||
}
|
||||
|
||||
// _ = context to keep import in case helpers above stop using it.
|
||||
var _ = context.Background
|
||||
@@ -0,0 +1,536 @@
|
||||
// Package scheduler — cel.go implements a minimal CEL-subset evaluator
|
||||
// for the CLI-side scheduler constraint expressions (REQ-083, P05).
|
||||
//
|
||||
// The full CEL specification (google.golang.org/genproto/...
|
||||
// googleapis/api/expr/v1alpha1) is intentionally NOT a dependency of
|
||||
// this module (see go.mod): adding it for a single callsite would pull
|
||||
// in a large transitive graph and contradict the "stdlib + minimal
|
||||
// deps" guardrail. Instead this file implements a hand-rolled
|
||||
// recursive-descent evaluator for the subset the PRD exercises:
|
||||
//
|
||||
// - attribute access on a `node.<name>` object (hostname, kind,
|
||||
// cpus, memory, tags, runtimes)
|
||||
// - string and integer literals (double-quoted)
|
||||
// - comparison operators: == != >= <= > <
|
||||
// - membership: <expr> in <expr>, <expr> not in <expr>
|
||||
// - boolean composition: and, or, not (parenthesised)
|
||||
//
|
||||
// Anything outside this subset returns an error rather than a silent
|
||||
// wrong answer; that is the documented limitation. The grammar is
|
||||
// small enough to be unambiguous with a top-down precedence-climbing
|
||||
// parser.
|
||||
package scheduler
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strconv"
|
||||
"strings"
|
||||
"unicode"
|
||||
)
|
||||
|
||||
// EvaluateConstraint evaluates a single CEL-subset expression against
|
||||
// the supplied NodeInfo. Returns (matched, err). An expression that
|
||||
// references an unknown attribute, uses an unsupported operator, or
|
||||
// fails to parse yields an error. Schedule treats a constraint
|
||||
// evaluation error as a non-fit (the node is silently skipped) rather
|
||||
// than a hard fail because operators routinely write exploratory
|
||||
// constraints against attributes the local cluster does not expose.
|
||||
func EvaluateConstraint(expr string, node NodeInfo) (bool, error) {
|
||||
p := newParser(strings.TrimSpace(expr), node)
|
||||
if p.len() == 0 {
|
||||
return false, fmt.Errorf("cel: empty expression")
|
||||
}
|
||||
v, err := p.parseExpr()
|
||||
if err != nil {
|
||||
return false, err
|
||||
}
|
||||
if p.tok.kind != tokEOF {
|
||||
return false, fmt.Errorf("cel: trailing input near %q", p.tok.text)
|
||||
}
|
||||
b, ok := v.(bool)
|
||||
if !ok {
|
||||
return false, fmt.Errorf("cel: expression did not evaluate to bool (got %T)", v)
|
||||
}
|
||||
return b, nil
|
||||
}
|
||||
|
||||
// EvaluateAll returns true iff every constraint evaluates to true
|
||||
// against the node (logical AND). An empty constraint list is vacuously
|
||||
// true. The first evaluation error short-circuits and is returned.
|
||||
func EvaluateAll(constraints []string, node NodeInfo) (bool, error) {
|
||||
for _, c := range constraints {
|
||||
ok, err := EvaluateConstraint(c, node)
|
||||
if err != nil {
|
||||
return false, fmt.Errorf("constraint %q: %w", c, err)
|
||||
}
|
||||
if !ok {
|
||||
return false, nil
|
||||
}
|
||||
}
|
||||
return true, nil
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Value model
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
// celValue is the union of values the evaluator produces. We use the
|
||||
// Go interface{} representation so that comparisons can be polymorphic
|
||||
// without a tagged-union ceremony; the supported concrete types are
|
||||
// bool, int64, and string. Lists are []celValue of the above.
|
||||
type celValue = interface{}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Tokenizer
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
type tokKind int
|
||||
|
||||
const (
|
||||
tokEOF tokKind = iota
|
||||
tokIdent
|
||||
tokInt
|
||||
tokStr
|
||||
tokOp // ==, !=, >=, <=, >, <, (, ), .
|
||||
tokIn // "in"
|
||||
tokAnd // "and"
|
||||
tokOr // "or"
|
||||
tokNot // "not"
|
||||
)
|
||||
|
||||
type token struct {
|
||||
kind tokKind
|
||||
text string
|
||||
}
|
||||
|
||||
type lexer struct {
|
||||
src string
|
||||
pos int
|
||||
}
|
||||
|
||||
func (l *lexer) next() (token, error) {
|
||||
for l.pos < len(l.src) && unicode.IsSpace(rune(l.src[l.pos])) {
|
||||
l.pos++
|
||||
}
|
||||
if l.pos >= len(l.src) {
|
||||
return token{kind: tokEOF}, nil
|
||||
}
|
||||
c := l.src[l.pos]
|
||||
// string literal
|
||||
if c == '"' {
|
||||
start := l.pos
|
||||
l.pos++
|
||||
for l.pos < len(l.src) && l.src[l.pos] != '"' {
|
||||
l.pos++
|
||||
}
|
||||
if l.pos >= len(l.src) {
|
||||
return token{}, fmt.Errorf("cel: unterminated string at %d", start)
|
||||
}
|
||||
val := l.src[start+1 : l.pos]
|
||||
l.pos++ // consume closing quote
|
||||
return token{kind: tokStr, text: val}, nil
|
||||
}
|
||||
// integer literal
|
||||
if unicode.IsDigit(rune(c)) {
|
||||
start := l.pos
|
||||
for l.pos < len(l.src) && unicode.IsDigit(rune(l.src[l.pos])) {
|
||||
l.pos++
|
||||
}
|
||||
return token{kind: tokInt, text: l.src[start:l.pos]}, nil
|
||||
}
|
||||
// identifier / keyword
|
||||
if isIdentStart(c) {
|
||||
start := l.pos
|
||||
for l.pos < len(l.src) && isIdentPart(l.src[l.pos]) {
|
||||
l.pos++
|
||||
}
|
||||
word := l.src[start:l.pos]
|
||||
switch word {
|
||||
case "in":
|
||||
return token{kind: tokIn, text: word}, nil
|
||||
case "and":
|
||||
return token{kind: tokAnd, text: word}, nil
|
||||
case "or":
|
||||
return token{kind: tokOr, text: word}, nil
|
||||
case "not":
|
||||
return token{kind: tokNot, text: word}, nil
|
||||
default:
|
||||
return token{kind: tokIdent, text: word}, nil
|
||||
}
|
||||
}
|
||||
// operators
|
||||
if strings.ContainsRune("()=!<>.", rune(c)) {
|
||||
// multi-char operators
|
||||
if l.pos+1 < len(l.src) {
|
||||
two := l.src[l.pos : l.pos+2]
|
||||
switch two {
|
||||
case "==", "!=", ">=", "<=":
|
||||
l.pos += 2
|
||||
return token{kind: tokOp, text: two}, nil
|
||||
}
|
||||
}
|
||||
l.pos++
|
||||
return token{kind: tokOp, text: string(c)}, nil
|
||||
}
|
||||
return token{}, fmt.Errorf("cel: unexpected character %q at %d", c, l.pos)
|
||||
}
|
||||
|
||||
func isIdentStart(c byte) bool {
|
||||
return c == '_' || (c >= 'a' && c <= 'z') || (c >= 'A' && c <= 'Z')
|
||||
}
|
||||
|
||||
func isIdentPart(c byte) bool {
|
||||
return isIdentStart(c) || (c >= '0' && c <= '9')
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Parser (recursive descent, precedence climbing)
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
type parser struct {
|
||||
src string
|
||||
pos int
|
||||
tok token
|
||||
err error
|
||||
node NodeInfo
|
||||
}
|
||||
|
||||
func newParser(src string, node NodeInfo) *parser {
|
||||
p := &parser{src: src, node: node}
|
||||
p.advance()
|
||||
return p
|
||||
}
|
||||
|
||||
func (p *parser) len() int { return len(p.src) }
|
||||
|
||||
func (p *parser) advance() {
|
||||
if p.err != nil {
|
||||
return
|
||||
}
|
||||
l := lexer{src: p.src, pos: p.pos}
|
||||
t, err := l.next()
|
||||
if err != nil {
|
||||
p.err = err
|
||||
return
|
||||
}
|
||||
p.pos = l.pos
|
||||
p.tok = t
|
||||
}
|
||||
|
||||
// Grammar (lowest precedence first):
|
||||
//
|
||||
// expr := orExpr
|
||||
// orExpr := andExpr ("or" andExpr)*
|
||||
// andExpr := notExpr ("and" notExpr)*
|
||||
// notExpr := "not" notExpr | cmpExpr
|
||||
// cmpExpr := primary (op primary | "in" primary | "not" "in" primary)?
|
||||
// primary := "(" expr ")"
|
||||
// | int
|
||||
// | str
|
||||
// | "true" | "false"
|
||||
// | nodeAttr ("." ident)? // node.<field>
|
||||
// | ident // bare attribute (e.g. region)
|
||||
// nodeAttr := "node"
|
||||
|
||||
func (p *parser) parseExpr() (celValue, error) {
|
||||
if p.err != nil {
|
||||
return nil, p.err
|
||||
}
|
||||
return p.parseOr()
|
||||
}
|
||||
|
||||
func (p *parser) parseOr() (celValue, error) {
|
||||
left, err := p.parseAnd()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
for p.tok.kind == tokOr {
|
||||
p.advance()
|
||||
right, err := p.parseAnd()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
lb, ok := left.(bool)
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("cel: 'or' operand not bool: %T", left)
|
||||
}
|
||||
rb, ok := right.(bool)
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("cel: 'or' operand not bool: %T", right)
|
||||
}
|
||||
left = lb || rb
|
||||
}
|
||||
return left, nil
|
||||
}
|
||||
|
||||
func (p *parser) parseAnd() (celValue, error) {
|
||||
left, err := p.parseNot()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
for p.tok.kind == tokAnd {
|
||||
p.advance()
|
||||
right, err := p.parseNot()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
lb, ok := left.(bool)
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("cel: 'and' operand not bool: %T", left)
|
||||
}
|
||||
rb, ok := right.(bool)
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("cel: 'and' operand not bool: %T", right)
|
||||
}
|
||||
left = lb && rb
|
||||
}
|
||||
return left, nil
|
||||
}
|
||||
|
||||
func (p *parser) parseNot() (celValue, error) {
|
||||
if p.tok.kind == tokNot {
|
||||
// "not" at the start of a primary is logical negation. "not in"
|
||||
// is handled in parseCmp where it follows a primary.
|
||||
p.advance()
|
||||
v, err := p.parseNot()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
b, ok := v.(bool)
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("cel: 'not' operand not bool: %T", v)
|
||||
}
|
||||
return !b, nil
|
||||
}
|
||||
return p.parseCmp()
|
||||
}
|
||||
|
||||
func (p *parser) parseCmp() (celValue, error) {
|
||||
left, err := p.parsePrimary()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
// "not in"
|
||||
if p.tok.kind == tokNot {
|
||||
p.advance()
|
||||
if p.tok.kind != tokIn {
|
||||
return nil, fmt.Errorf("cel: expected 'in' after 'not', got %q", p.tok.text)
|
||||
}
|
||||
p.advance()
|
||||
right, err := p.parsePrimary()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
member, err := inMember(left, right)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return !member, nil
|
||||
}
|
||||
// "in"
|
||||
if p.tok.kind == tokIn {
|
||||
p.advance()
|
||||
right, err := p.parsePrimary()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return inMember(left, right)
|
||||
}
|
||||
// comparison operators
|
||||
if p.tok.kind == tokOp {
|
||||
op := p.tok.text
|
||||
switch op {
|
||||
case "==", "!=", ">=", "<=", ">", "<":
|
||||
p.advance()
|
||||
right, err := p.parsePrimary()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return compare(op, left, right)
|
||||
default:
|
||||
return nil, fmt.Errorf("cel: unexpected operator %q", op)
|
||||
}
|
||||
}
|
||||
return left, nil
|
||||
}
|
||||
|
||||
// inMember reports whether left is a member of right. right must be a
|
||||
// list ([]celValue) of comparable values; left may be a string or
|
||||
// int64.
|
||||
func inMember(left, right celValue) (bool, error) {
|
||||
list, ok := right.([]celValue)
|
||||
if !ok {
|
||||
return false, fmt.Errorf("cel: 'in' rhs not a list: %T", right)
|
||||
}
|
||||
for _, e := range list {
|
||||
if valuesEqual(left, e) {
|
||||
return true, nil
|
||||
}
|
||||
}
|
||||
return false, nil
|
||||
}
|
||||
|
||||
func valuesEqual(a, b celValue) bool {
|
||||
switch av := a.(type) {
|
||||
case string:
|
||||
bv, ok := b.(string)
|
||||
return ok && av == bv
|
||||
case int64:
|
||||
bv, ok := b.(int64)
|
||||
return ok && av == bv
|
||||
case bool:
|
||||
bv, ok := b.(bool)
|
||||
return ok && av == bv
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// compare applies a binary comparison operator to two scalar values.
|
||||
// Strings compare lexicographically; ints numerically; bools only via
|
||||
// ==/!=.
|
||||
func compare(op string, left, right celValue) (bool, error) {
|
||||
switch op {
|
||||
case "==":
|
||||
return valuesEqual(left, right), nil
|
||||
case "!=":
|
||||
return !valuesEqual(left, right), nil
|
||||
}
|
||||
// ordered comparisons require ordered operands
|
||||
ls, lok := left.(string)
|
||||
rs, rok := right.(string)
|
||||
if lok && rok {
|
||||
switch op {
|
||||
case "<":
|
||||
return ls < rs, nil
|
||||
case "<=":
|
||||
return ls <= rs, nil
|
||||
case ">":
|
||||
return ls > rs, nil
|
||||
case ">=":
|
||||
return ls >= rs, nil
|
||||
}
|
||||
}
|
||||
li, lok := left.(int64)
|
||||
ri, rok := right.(int64)
|
||||
if lok && rok {
|
||||
switch op {
|
||||
case "<":
|
||||
return li < ri, nil
|
||||
case "<=":
|
||||
return li <= ri, nil
|
||||
case ">":
|
||||
return li > ri, nil
|
||||
case ">=":
|
||||
return li >= ri, nil
|
||||
}
|
||||
}
|
||||
return false, fmt.Errorf("cel: cannot apply %q to %T and %T", op, left, right)
|
||||
}
|
||||
|
||||
// parsePrimary parses the smallest standalone unit: parenthesised
|
||||
// expressions, literals, and attribute references.
|
||||
func (p *parser) parsePrimary() (celValue, error) {
|
||||
switch p.tok.kind {
|
||||
case tokOp:
|
||||
if p.tok.text == "(" {
|
||||
p.advance()
|
||||
v, err := p.parseExpr()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if p.tok.kind != tokOp || p.tok.text != ")" {
|
||||
return nil, fmt.Errorf("cel: expected ')' got %q", p.tok.text)
|
||||
}
|
||||
p.advance()
|
||||
return v, nil
|
||||
}
|
||||
return nil, fmt.Errorf("cel: unexpected operator %q", p.tok.text)
|
||||
case tokInt:
|
||||
n, err := strconv.ParseInt(p.tok.text, 10, 64)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("cel: bad int %q: %w", p.tok.text, err)
|
||||
}
|
||||
p.advance()
|
||||
return n, nil
|
||||
case tokStr:
|
||||
v := p.tok.text
|
||||
p.advance()
|
||||
return v, nil
|
||||
case tokIdent:
|
||||
return p.parseAttrRef()
|
||||
}
|
||||
return nil, fmt.Errorf("cel: unexpected token %q", p.tok.text)
|
||||
}
|
||||
|
||||
// parseAttrRef resolves a bare or `node.<field>` attribute reference
|
||||
// against the node being evaluated. Bare identifiers (e.g. `region`)
|
||||
// resolve against the same attribute map as `node.region`; the PRD
|
||||
// examples use both forms interchangeably (see
|
||||
// TestParseMarkdown_ConstraintsInlineArray).
|
||||
func (p *parser) parseAttrRef() (celValue, error) {
|
||||
name := p.tok.text
|
||||
p.advance()
|
||||
// dotted access: node.<field>
|
||||
if p.tok.kind == tokOp && p.tok.text == "." {
|
||||
if name != "node" {
|
||||
return nil, fmt.Errorf("cel: dotted access on non-node: %q", name)
|
||||
}
|
||||
p.advance()
|
||||
if p.tok.kind != tokIdent {
|
||||
return nil, fmt.Errorf("cel: expected attribute name after '.', got %q", p.tok.text)
|
||||
}
|
||||
field := p.tok.text
|
||||
p.advance()
|
||||
return p.nodeAttr(name + "." + field)
|
||||
}
|
||||
// bare identifier
|
||||
switch name {
|
||||
case "true":
|
||||
return true, nil
|
||||
case "false":
|
||||
return false, nil
|
||||
default:
|
||||
return p.nodeAttr(name)
|
||||
}
|
||||
}
|
||||
|
||||
// nodeAttr resolves an attribute name to its value on the parser's
|
||||
// active node. Mapping (per PRD T2):
|
||||
//
|
||||
// node.hostname -> Hostname (string)
|
||||
// node.kind -> Kind (string)
|
||||
// node.cpus -> CPU (int64)
|
||||
// node.memory -> Memory (int64)
|
||||
// node.tags -> Tags ([]string -> []celValue)
|
||||
// node.runtimes -> Runtimes ([]string -> []celValue)
|
||||
//
|
||||
// Bare names (without the `node.` prefix) resolve through the same
|
||||
// map, so `region == "us"` and `node.region == "us"` are equivalent
|
||||
// when the attribute exists.
|
||||
func (p *parser) nodeAttr(name string) (celValue, error) {
|
||||
switch name {
|
||||
case "node.hostname", "hostname":
|
||||
return p.node.Hostname, nil
|
||||
case "node.kind", "kind":
|
||||
return p.node.Kind, nil
|
||||
case "node.cpus", "cpus":
|
||||
return p.node.CPU, nil
|
||||
case "node.memory", "memory":
|
||||
return p.node.Memory, nil
|
||||
case "node.tags", "tags":
|
||||
return toStringValues(p.node.Tags), nil
|
||||
case "node.runtimes", "runtimes":
|
||||
return toStringValues(p.node.Runtimes), nil
|
||||
}
|
||||
return nil, fmt.Errorf("cel: unknown attribute %q", name)
|
||||
}
|
||||
|
||||
// toStringValues converts a []string to []celValue so the membership
|
||||
// operators can compare element-wise.
|
||||
func toStringValues(in []string) []celValue {
|
||||
out := make([]celValue, len(in))
|
||||
for i, s := range in {
|
||||
out[i] = s
|
||||
}
|
||||
return out
|
||||
}
|
||||
@@ -0,0 +1,210 @@
|
||||
package scheduler
|
||||
|
||||
import "testing"
|
||||
|
||||
func TestEvaluateConstraint_Equality(t *testing.T) {
|
||||
node := NodeInfo{Hostname: "h-1", Kind: "linux", CPU: 4, Memory: 4096, Tags: []string{"web"}, Runtimes: []string{"process"}}
|
||||
cases := []struct {
|
||||
name string
|
||||
expr string
|
||||
want bool
|
||||
}{
|
||||
{"hostname eq", `node.hostname == "h-1"`, true},
|
||||
{"hostname ne", `node.hostname == "h-2"`, false},
|
||||
{"kind eq", `node.kind == "linux"`, true},
|
||||
{"kind ne", `node.kind == "proxmox"`, false},
|
||||
{"cpus eq", `node.cpus == 4`, true},
|
||||
{"memory eq", `node.memory == 4096`, true},
|
||||
}
|
||||
for _, c := range cases {
|
||||
got, err := EvaluateConstraint(c.expr, node)
|
||||
if err != nil {
|
||||
t.Errorf("%s: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if got != c.want {
|
||||
t.Errorf("%s: got %v, want %v", c.name, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestEvaluateConstraint_Comparison(t *testing.T) {
|
||||
node := NodeInfo{Hostname: "h", Kind: "linux", CPU: 4, Memory: 4096}
|
||||
cases := []struct {
|
||||
expr string
|
||||
want bool
|
||||
}{
|
||||
{"node.cpus >= 2", true},
|
||||
{"node.cpus >= 4", true},
|
||||
{"node.cpus > 4", false},
|
||||
{"node.cpus > 2", true},
|
||||
{"node.cpus <= 4", true},
|
||||
{"node.cpus < 2", false},
|
||||
{"node.cpus != 8", true},
|
||||
{"node.cpus == 8", false},
|
||||
{"node.memory >= 2048", true},
|
||||
{"node.memory < 1024", false},
|
||||
}
|
||||
for _, c := range cases {
|
||||
got, err := EvaluateConstraint(c.expr, node)
|
||||
if err != nil {
|
||||
t.Errorf("%q: %v", c.expr, err)
|
||||
continue
|
||||
}
|
||||
if got != c.want {
|
||||
t.Errorf("%q: got %v, want %v", c.expr, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestEvaluateConstraint_Membership(t *testing.T) {
|
||||
node := NodeInfo{Tags: []string{"web", "log-shipper"}, Runtimes: []string{"process", "wasmtime"}}
|
||||
cases := []struct {
|
||||
expr string
|
||||
want bool
|
||||
}{
|
||||
{`"web" in node.tags`, true},
|
||||
{`"missing" in node.tags`, false},
|
||||
{`"process" in node.runtimes`, true},
|
||||
{`"podman" in node.runtimes`, false},
|
||||
{`"log-shipper" not in node.tags`, false},
|
||||
{`"missing" not in node.tags`, true},
|
||||
}
|
||||
for _, c := range cases {
|
||||
got, err := EvaluateConstraint(c.expr, node)
|
||||
if err != nil {
|
||||
t.Errorf("%q: %v", c.expr, err)
|
||||
continue
|
||||
}
|
||||
if got != c.want {
|
||||
t.Errorf("%q: got %v, want %v", c.expr, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestEvaluateConstraint_BooleanComposition(t *testing.T) {
|
||||
node := NodeInfo{Kind: "linux", CPU: 4, Tags: []string{"web"}}
|
||||
cases := []struct {
|
||||
expr string
|
||||
want bool
|
||||
}{
|
||||
{`node.kind == "linux" and node.cpus >= 2`, true},
|
||||
{`node.kind == "proxmox" and node.cpus >= 2`, false},
|
||||
{`node.kind == "linux" or node.kind == "proxmox"`, true},
|
||||
{`node.kind == "proxmox" or node.kind == "linux"`, true},
|
||||
{`not node.kind == "proxmox"`, true},
|
||||
{`not node.kind == "linux"`, false},
|
||||
{`(node.kind == "linux") and (node.cpus >= 2)`, true},
|
||||
{`node.cpus >= 2 and not "blocked" in node.tags`, true},
|
||||
{`node.kind == "linux" and node.cpus >= 2 and "web" in node.tags`, true},
|
||||
{`node.kind == "linux" or node.kind == "proxmox" or node.cpus > 100`, true},
|
||||
}
|
||||
for _, c := range cases {
|
||||
got, err := EvaluateConstraint(c.expr, node)
|
||||
if err != nil {
|
||||
t.Errorf("%q: %v", c.expr, err)
|
||||
continue
|
||||
}
|
||||
if got != c.want {
|
||||
t.Errorf("%q: got %v, want %v", c.expr, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestEvaluateConstraint_BareIdentifiers(t *testing.T) {
|
||||
// Bare identifiers resolve through the same attribute map as
|
||||
// node.<field> (per PRD: constraints may use either form).
|
||||
node := NodeInfo{Kind: "linux", CPU: 4}
|
||||
got, err := EvaluateConstraint(`kind == "linux"`, node)
|
||||
if err != nil {
|
||||
t.Fatalf("bare kind: %v", err)
|
||||
}
|
||||
if !got {
|
||||
t.Error("bare kind == linux: got false, want true")
|
||||
}
|
||||
}
|
||||
|
||||
func TestEvaluateConstraint_TrueFalseLiterals(t *testing.T) {
|
||||
node := NodeInfo{}
|
||||
cases := []struct {
|
||||
expr string
|
||||
want bool
|
||||
}{
|
||||
{"true", true},
|
||||
{"false", false},
|
||||
{"not false", true},
|
||||
{"not true", false},
|
||||
{"true and true", true},
|
||||
{"true and false", false},
|
||||
{"false or true", true},
|
||||
}
|
||||
for _, c := range cases {
|
||||
got, err := EvaluateConstraint(c.expr, node)
|
||||
if err != nil {
|
||||
t.Errorf("%q: %v", c.expr, err)
|
||||
continue
|
||||
}
|
||||
if got != c.want {
|
||||
t.Errorf("%q: got %v, want %v", c.expr, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestEvaluateConstraint_Errors(t *testing.T) {
|
||||
node := NodeInfo{Kind: "linux"}
|
||||
cases := []struct {
|
||||
name string
|
||||
expr string
|
||||
}{
|
||||
{"empty", ""},
|
||||
{"unterminated string", `node.kind == "linux`},
|
||||
{"unknown attribute", `node.bogus == 1`},
|
||||
{"unknown bare attr", `bogus == 1`},
|
||||
{"dotted on non-node", `host.kind == "linux"`},
|
||||
{"bad operator", `node.cpus + 2`},
|
||||
{"trailing input", `node.kind == "linux" garbage`},
|
||||
{"unbalanced paren", `(node.kind == "linux"`},
|
||||
{"missing rhs", `node.cpus >=`},
|
||||
{"not without in", `"x" not node.tags`},
|
||||
{"ordered compare on bool", `true < false`},
|
||||
{"ordered compare on mismatched types", `node.kind > 2`},
|
||||
{"in on non-list", `"x" in node.kind`},
|
||||
}
|
||||
for _, c := range cases {
|
||||
_, err := EvaluateConstraint(c.expr, node)
|
||||
if err == nil {
|
||||
t.Errorf("%s: expected error for %q, got nil", c.name, c.expr)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestEvaluateAll(t *testing.T) {
|
||||
node := NodeInfo{Kind: "linux", CPU: 4, Tags: []string{"web"}}
|
||||
cases := []struct {
|
||||
name string
|
||||
constraints []string
|
||||
want bool
|
||||
}{
|
||||
{"empty", nil, true},
|
||||
{"all pass", []string{`node.kind == "linux"`, "node.cpus >= 2"}, true},
|
||||
{"one fails", []string{`node.kind == "linux"`, "node.cpus >= 8"}, false},
|
||||
{"all fail", []string{`node.kind == "proxmox"`, "node.cpus >= 8"}, false},
|
||||
}
|
||||
for _, c := range cases {
|
||||
got, err := EvaluateAll(c.constraints, node)
|
||||
if err != nil {
|
||||
t.Errorf("%s: %v", c.name, err)
|
||||
continue
|
||||
}
|
||||
if got != c.want {
|
||||
t.Errorf("%s: got %v, want %v", c.name, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestEvaluateAll_PropagatesError(t *testing.T) {
|
||||
node := NodeInfo{}
|
||||
if _, err := EvaluateAll([]string{"bogus == 1"}, node); err == nil {
|
||||
t.Error("EvaluateAll: expected error for malformed constraint")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,472 @@
|
||||
// Package scheduler implements the v0.9 CLI-side scheduler (REQ-083,
|
||||
// P05). Unlike the v0.8 daemon-side best-fit scheduler
|
||||
// (internal/engine/scheduler.go), this scheduler runs entirely in the
|
||||
// `orca` CLI process (R-001) and is pure: it takes a list of candidate
|
||||
// nodes plus a workload request and returns placement decisions
|
||||
// without performing any I/O.
|
||||
//
|
||||
// The scheduler is runtime-aware: a workload that declares
|
||||
// `runtime.one_of: wasm` is only placed on nodes that expose
|
||||
// `wasmtime` in their Runtimes list; a `pve-vm` workload is only
|
||||
// placed on `proxmox` nodes. It is also constraint- and
|
||||
// affinity-aware via the CEL-subset evaluator in cel.go.
|
||||
//
|
||||
// Workload kinds are handled differently per the PRD:
|
||||
//
|
||||
// - Job: one-shot, returns exactly one placement (best-fit
|
||||
// bin-packing).
|
||||
// - Service: count replicas spread across distinct nodes
|
||||
// (anti-affinity by default); if fewer distinct nodes than count,
|
||||
// colocation is permitted but distinct nodes are preferred.
|
||||
// - DaemonSet: one placement per node that fits the constraints.
|
||||
package scheduler
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"sort"
|
||||
|
||||
"git.cloudinit.dev/coreci/orca/internal/jobspec"
|
||||
)
|
||||
|
||||
// NodeInfo is the scheduler's projection of a peer node: total and
|
||||
// free capacity, the runtimes the node advertises, its tags, and its
|
||||
// kind (linux/proxmox). The CLI populates this from the
|
||||
// cluster/peers/ inventory plus the per-node capacity reports
|
||||
// collected over SSH; the scheduler itself never reads either.
|
||||
type NodeInfo struct {
|
||||
Hostname string
|
||||
Runtimes []string
|
||||
Tags []string
|
||||
CPU int64
|
||||
Memory int64
|
||||
FreeCPU int64
|
||||
FreeMem int64
|
||||
Kind string
|
||||
}
|
||||
|
||||
// WorkloadRequest bundles a parsed WorkloadSpec with the namespace
|
||||
// the workload is being scheduled into. The namespace is carried
|
||||
// through to placement so the resulting AllocID can be namespaced,
|
||||
// but the scheduler itself does not inspect it for fitting decisions.
|
||||
type WorkloadRequest struct {
|
||||
Spec *jobspec.WorkloadSpec
|
||||
Namespace string
|
||||
}
|
||||
|
||||
// Placement is a single scheduling decision: which node, which
|
||||
// allocation id, and the bin-packing score that won the node the
|
||||
// placement. AllocID is `ns/spec.Name-<idx>` so a multi-replica
|
||||
// Service produces distinct ids per replica.
|
||||
type Placement struct {
|
||||
Node string
|
||||
AllocID string
|
||||
Score int64
|
||||
}
|
||||
|
||||
// Schedule is the main entry point. For Job (kind=Job) it returns one
|
||||
// placement on the best-fit node. For Service it returns `Count`
|
||||
// placements spread across distinct nodes where possible (anti-
|
||||
// affinity), permitting colocation when Count > nodes. For DaemonSet
|
||||
// it returns one placement per node that fits. Any kind-agnostic
|
||||
// validation error (no spec, unknown kind, no fitting node) is
|
||||
// returned as an error rather than an empty slice so callers can
|
||||
// distinguish "nothing fits" from "scheduled zero replicas".
|
||||
func Schedule(nodes []NodeInfo, req WorkloadRequest) ([]Placement, error) {
|
||||
if req.Spec == nil {
|
||||
return nil, fmt.Errorf("scheduler: nil WorkloadSpec")
|
||||
}
|
||||
if len(nodes) == 0 {
|
||||
return nil, fmt.Errorf("scheduler: no candidate nodes")
|
||||
}
|
||||
switch req.Spec.Kind {
|
||||
case "Job":
|
||||
return scheduleJob(nodes, req)
|
||||
case "Service":
|
||||
return scheduleService(nodes, req)
|
||||
case "DaemonSet":
|
||||
return scheduleDaemonSet(nodes, req)
|
||||
default:
|
||||
return nil, fmt.Errorf("scheduler: unknown kind %q", req.Spec.Kind)
|
||||
}
|
||||
}
|
||||
|
||||
// Score evaluates a single node against a workload. fits is true iff
|
||||
// the node (a) advertises a runtime compatible with the workload's
|
||||
// `runtime.one_of`, (b) satisfies every CEL constraint in
|
||||
// `spec.Constraints`, and (c) has enough free CPU+memory for the
|
||||
// workload's requested resources. When fits is true, score is the
|
||||
// bin-packing score (more free capacity = higher score, so the node
|
||||
// most likely to absorb the workload without starving its
|
||||
// neighbours wins). When fits is false, score is 0.
|
||||
func Score(node NodeInfo, req WorkloadRequest) (score int64, fits bool) {
|
||||
// (a) runtime compatibility. A workload with no Runtime block or
|
||||
// an empty OneOf is treated as runtime-agnostic (always fits on
|
||||
// the runtime axis); this matches the v0.8 behaviour where a
|
||||
// missing runtime meant "process".
|
||||
runtimeOK := true
|
||||
if req.Spec != nil && req.Spec.Runtime != nil && req.Spec.Runtime.OneOf != "" {
|
||||
runtimeOK = hasRuntime(node, req.Spec.Runtime.OneOf)
|
||||
}
|
||||
if !runtimeOK {
|
||||
return 0, false
|
||||
}
|
||||
// (b) constraints. Evaluation errors are treated as non-fit so a
|
||||
// malformed constraint does not crash Schedule; the caller still
|
||||
// sees the node filtered out.
|
||||
if req.Spec != nil {
|
||||
ok, err := EvaluateAll(req.Spec.Constraints, node)
|
||||
if err != nil || !ok {
|
||||
return 0, false
|
||||
}
|
||||
}
|
||||
// (c) capacity. A workload with no Resources block is treated as
|
||||
// zero-sized for fitting purposes (it always fits the capacity
|
||||
// axis); real workloads declare cpu/memory.
|
||||
needCPU, needMem := workloadResources(req)
|
||||
if node.FreeCPU < needCPU || node.FreeMem < needMem {
|
||||
return 0, false
|
||||
}
|
||||
// bin-packing score: most free capacity wins. CPU is weighted
|
||||
// 1000x memory so a 1-core difference outweighs a 1-MiB
|
||||
// difference, mirroring the v0.8 Score weighting that biased
|
||||
// toward CPU (the more common binding constraint).
|
||||
score = (node.FreeCPU-needCPU)*1000 + (node.FreeMem - needMem)
|
||||
if score < 0 {
|
||||
score = 0
|
||||
}
|
||||
return score, true
|
||||
}
|
||||
|
||||
// hasRuntime reports whether node advertises the requested runtime.
|
||||
// The match is case-insensitive and tolerant of aliases: `wasm` and
|
||||
// `wasmtime` are treated as the same runtime, and `pve-vm`/`pve-ct`
|
||||
// only match nodes whose Kind is "proxmox".
|
||||
func hasRuntime(node NodeInfo, oneOf string) bool {
|
||||
want := normalizeRuntime(oneOf)
|
||||
// pve-* runtimes require a proxmox-kind node regardless of the
|
||||
// node's Runtimes list (a proxmox node doesn't list "pve-vm" in
|
||||
// Runtimes; it IS the runtime).
|
||||
switch want {
|
||||
case "pve-vm", "pve-ct", "proxmox":
|
||||
return normalizeKind(node.Kind) == "proxmox"
|
||||
}
|
||||
for _, r := range node.Runtimes {
|
||||
if normalizeRuntime(r) == want {
|
||||
return true
|
||||
}
|
||||
// alias: wasmtime nodes advertise "wasmtime"; workloads ask
|
||||
// for "wasm".
|
||||
if want == "wasm" && normalizeRuntime(r) == "wasmtime" {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// normalizeRuntime lowercases and trims a runtime name for matching.
|
||||
func normalizeRuntime(s string) string {
|
||||
s = toLowerASCII(s)
|
||||
switch s {
|
||||
case "wasmtime":
|
||||
return "wasm"
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
// normalizeKind lowercases and trims a node Kind for matching.
|
||||
func normalizeKind(s string) string { return toLowerASCII(s) }
|
||||
|
||||
// toLowerASCII lowercases ASCII letters without bringing in strings
|
||||
// (avoid an alloc-heavy stdlib call in the hot path).
|
||||
func toLowerASCII(s string) string {
|
||||
b := []byte(s)
|
||||
for i, c := range b {
|
||||
if c >= 'A' && c <= 'Z' {
|
||||
b[i] = c + 32
|
||||
}
|
||||
}
|
||||
return string(b)
|
||||
}
|
||||
|
||||
// workloadResources returns the (cpu, memory) the workload requests,
|
||||
// read from the WorkloadSpec's Resources block if present. The
|
||||
// v0.9-P05 WorkloadSpec does not yet carry a Resources field (it
|
||||
// lands in P0c, REQ-074); until then this returns (0, 0) so the
|
||||
// capacity check is a no-op and runtime/constraints do the real
|
||||
// filtering. The signature is here so the scheduler logic does not
|
||||
// need to change when Resources lands.
|
||||
func workloadResources(req WorkloadRequest) (int64, int64) {
|
||||
_ = req
|
||||
return 0, 0
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Kind-specific scheduling
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
// scheduleJob places a single Job on the best-fit node.
|
||||
func scheduleJob(nodes []NodeInfo, req WorkloadRequest) ([]Placement, error) {
|
||||
type cand struct {
|
||||
node NodeInfo
|
||||
score int64
|
||||
}
|
||||
var cands []cand
|
||||
for _, n := range nodes {
|
||||
s, ok := Score(n, req)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
cands = append(cands, cand{node: n, score: s})
|
||||
}
|
||||
// CEL-based affinity rules apply to single-shot Jobs too: a Job
|
||||
// with `affinity: [{target: "\"ssd\" in node.tags", weight: 100}]`
|
||||
// should land on the tagged node even without prior placements.
|
||||
// Name-based affinity (no prior placements to check) contributes
|
||||
// zero for a standalone Job, so it is harmless to call here.
|
||||
for i := range cands {
|
||||
cands[i].score += affinityScore(cands[i].node, req, nil)
|
||||
}
|
||||
if len(cands) == 0 {
|
||||
return nil, fmt.Errorf("scheduler: no node fits workload %q", req.Spec.Name)
|
||||
}
|
||||
sort.SliceStable(cands, func(i, j int) bool {
|
||||
if cands[i].score != cands[j].score {
|
||||
return cands[i].score > cands[j].score
|
||||
}
|
||||
return cands[i].node.Hostname < cands[j].node.Hostname
|
||||
})
|
||||
w := cands[0]
|
||||
return []Placement{{
|
||||
Node: w.node.Hostname,
|
||||
AllocID: allocID(req, 0),
|
||||
Score: w.score,
|
||||
}}, nil
|
||||
}
|
||||
|
||||
// scheduleService places `Count` replicas with implicit anti-affinity:
|
||||
// prefer distinct nodes, but permit colocation when Count exceeds the
|
||||
// number of fitting nodes. Each replica gets a distinct AllocID.
|
||||
func scheduleService(nodes []NodeInfo, req WorkloadRequest) ([]Placement, error) {
|
||||
count := req.Spec.Count
|
||||
if count <= 0 {
|
||||
count = 1
|
||||
}
|
||||
// Pre-filter fitting nodes once; the loop below re-scores them
|
||||
// after each placement so the capacity accounting reflects the
|
||||
// replicas already placed.
|
||||
fitting := filterFitting(nodes, req)
|
||||
if len(fitting) == 0 {
|
||||
return nil, fmt.Errorf("scheduler: no node fits service %q", req.Spec.Name)
|
||||
}
|
||||
var placements []Placement
|
||||
placed := map[string]int{} // hostname -> count placed there
|
||||
// First pass: spread across distinct nodes.
|
||||
for i := 0; i < count; i++ {
|
||||
best, score, ok := pickServiceNode(fitting, req, placements, placed)
|
||||
if !ok {
|
||||
break
|
||||
}
|
||||
placements = append(placements, Placement{
|
||||
Node: best.Hostname,
|
||||
AllocID: allocID(req, i),
|
||||
Score: score,
|
||||
})
|
||||
placed[best.Hostname]++
|
||||
// Reflect the consumed capacity in the candidate snapshot so
|
||||
// subsequent picks see updated free capacity.
|
||||
needCPU, needMem := workloadResources(req)
|
||||
for j := range fitting {
|
||||
if fitting[j].Hostname == best.Hostname {
|
||||
fitting[j].FreeCPU -= needCPU
|
||||
fitting[j].FreeMem -= needMem
|
||||
}
|
||||
}
|
||||
}
|
||||
if len(placements) < count {
|
||||
return nil, fmt.Errorf("scheduler: only placed %d/%d replicas for service %q",
|
||||
len(placements), count, req.Spec.Name)
|
||||
}
|
||||
return placements, nil
|
||||
}
|
||||
|
||||
// pickServiceNode selects the best node for the next replica. The
|
||||
// selection prefers nodes with zero prior placements of this service
|
||||
// (anti-affinity) and applies affinity scoring on top of the
|
||||
// bin-packing score.
|
||||
func pickServiceNode(fitting []NodeInfo, req WorkloadRequest, placements []Placement, placed map[string]int) (NodeInfo, int64, bool) {
|
||||
type scored struct {
|
||||
node NodeInfo
|
||||
score int64
|
||||
}
|
||||
var cands []scored
|
||||
for _, n := range fitting {
|
||||
s, ok := Score(n, req)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
// Implicit anti-affinity: a node with N prior replicas of this
|
||||
// service incurs a penalty of N * (1 << 62) so distinct nodes
|
||||
// are preferred, but colocation is permitted (with a
|
||||
// per-replica penalty) when no distinct node remains. This
|
||||
// produces a balanced spread (e.g. 5 replicas on 3 nodes →
|
||||
// 2/2/1) rather than stacking everything on the first node.
|
||||
if placed[n.Hostname] > 0 {
|
||||
s -= int64(placed[n.Hostname]) * (1 << 62)
|
||||
}
|
||||
// Affinity rules from the spec add/subtract their weight.
|
||||
s += affinityScore(n, req, placements)
|
||||
cands = append(cands, scored{node: n, score: s})
|
||||
}
|
||||
if len(cands) == 0 {
|
||||
return NodeInfo{}, 0, false
|
||||
}
|
||||
sort.SliceStable(cands, func(i, j int) bool {
|
||||
if cands[i].score != cands[j].score {
|
||||
return cands[i].score > cands[j].score
|
||||
}
|
||||
return cands[i].node.Hostname < cands[j].node.Hostname
|
||||
})
|
||||
w := cands[0]
|
||||
return w.node, w.score, true
|
||||
}
|
||||
|
||||
// scheduleDaemonSet places one replica per node that fits the
|
||||
// constraints. The PRD's DaemonSet placement mode (every-node /
|
||||
// matching / mandatory) lives on the WorkloadSpec.Schedule block; the
|
||||
// scheduler honours it indirectly by filtering on Constraints: a
|
||||
// `matching` DaemonSet carries constraints that select the matching
|
||||
// nodes, an `every-node` DaemonSet carries none, and a `mandatory`
|
||||
// one is enforced elsewhere (the scheduler still just returns
|
||||
// placements for every fitting node).
|
||||
func scheduleDaemonSet(nodes []NodeInfo, req WorkloadRequest) ([]Placement, error) {
|
||||
var placements []Placement
|
||||
for _, n := range nodes {
|
||||
s, ok := Score(n, req)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
placements = append(placements, Placement{
|
||||
Node: n.Hostname,
|
||||
AllocID: allocID(req, len(placements)),
|
||||
Score: s,
|
||||
})
|
||||
}
|
||||
if len(placements) == 0 {
|
||||
return nil, fmt.Errorf("scheduler: no node fits daemonset %q", req.Spec.Name)
|
||||
}
|
||||
return placements, nil
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Affinity scoring
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
// affinityScore returns the weighted affinity contribution for a
|
||||
// node given the placements already made. For each AffinityRule the
|
||||
// Target is a CEL expression; if it evaluates true against the node,
|
||||
// the rule's Weight is added (positive = co-locate, negative =
|
||||
// anti-affinity). An affinity target that fails to evaluate is
|
||||
// ignored rather than failing the schedule: operators use affinity as
|
||||
// a hint, not a hard gate.
|
||||
//
|
||||
// The PRD also mentions affinity rules like `{target: "redis", weight:
|
||||
// 50}` where Target is a workload *name* rather than a CEL expression.
|
||||
// We support both: if Target parses as a CEL expression it is
|
||||
// evaluated against the node; otherwise it is treated as a workload
|
||||
// name and we check whether any already-placed alloc for that name
|
||||
// exists on the node. The placement-already-here check is done by the
|
||||
// caller via placements; this function checks the node's own
|
||||
// attributes only.
|
||||
func affinityScore(node NodeInfo, req WorkloadRequest, placements []Placement) int64 {
|
||||
if req.Spec == nil {
|
||||
return 0
|
||||
}
|
||||
var total int64
|
||||
for _, rule := range req.Spec.Affinity {
|
||||
// Try CEL evaluation first; if the target is a bare workload
|
||||
// name (no operator) the CEL parser will fail and we fall
|
||||
// back to name-based placement counting.
|
||||
ok, err := EvaluateConstraint(rule.Target, node)
|
||||
if err == nil {
|
||||
if ok {
|
||||
total += int64(rule.Weight)
|
||||
}
|
||||
continue
|
||||
}
|
||||
// Fallback: target is a workload name; count existing
|
||||
// placements for that workload on this node and apply the
|
||||
// weight once per co-located replica.
|
||||
for _, p := range placements {
|
||||
if p.Node == node.Hostname && isAllocFor(p.AllocID, rule.Target) {
|
||||
total += int64(rule.Weight)
|
||||
}
|
||||
}
|
||||
}
|
||||
return total
|
||||
}
|
||||
|
||||
// isAllocFor reports whether an AllocID encodes a placement for the
|
||||
// named workload. AllocIDs are `ns/name-idx`, so we look for the
|
||||
// workload name as the segment after the first slash and before the
|
||||
// trailing `-idx`.
|
||||
func isAllocFor(allocID, workloadName string) bool {
|
||||
// strip namespace prefix
|
||||
rest := allocID
|
||||
if i := indexByte(rest, '/'); i >= 0 {
|
||||
rest = rest[i+1:]
|
||||
}
|
||||
// strip trailing -idx
|
||||
if i := lastIndexByte(rest, '-'); i >= 0 {
|
||||
rest = rest[:i]
|
||||
}
|
||||
return rest == workloadName
|
||||
}
|
||||
|
||||
// indexByte returns the index of the first occurrence of b in s, or
|
||||
// -1. Avoids importing strings just for one helper.
|
||||
func indexByte(s string, b byte) int {
|
||||
for i := 0; i < len(s); i++ {
|
||||
if s[i] == b {
|
||||
return i
|
||||
}
|
||||
}
|
||||
return -1
|
||||
}
|
||||
|
||||
// lastIndexByte returns the index of the last occurrence of b in s, or
|
||||
// -1.
|
||||
func lastIndexByte(s string, b byte) int {
|
||||
for i := len(s) - 1; i >= 0; i-- {
|
||||
if s[i] == b {
|
||||
return i
|
||||
}
|
||||
}
|
||||
return -1
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Helpers
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
// filterFitting returns a copy of the nodes that pass Score for the
|
||||
// request, preserving order. Capacity is not yet decremented; the
|
||||
// caller adjusts FreeCPU/FreeMem as it places replicas.
|
||||
func filterFitting(nodes []NodeInfo, req WorkloadRequest) []NodeInfo {
|
||||
var out []NodeInfo
|
||||
for _, n := range nodes {
|
||||
if _, ok := Score(n, req); ok {
|
||||
out = append(out, n)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// allocID renders a stable, namespaced allocation id for a placement.
|
||||
// Format: `ns/spec.Name-<idx>`.
|
||||
func allocID(req WorkloadRequest, idx int) string {
|
||||
ns := req.Namespace
|
||||
if ns == "" {
|
||||
ns = "default"
|
||||
}
|
||||
return fmt.Sprintf("%s/%s-%d", ns, req.Spec.Name, idx)
|
||||
}
|
||||
@@ -0,0 +1,451 @@
|
||||
package scheduler
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"git.cloudinit.dev/coreci/orca/internal/jobspec"
|
||||
)
|
||||
|
||||
// threeLinuxNodes returns a small cluster of three Linux nodes with
|
||||
// distinct free capacities so best-fit ordering is unambiguous.
|
||||
func threeLinuxNodes() []NodeInfo {
|
||||
return []NodeInfo{
|
||||
{Hostname: "node-a", Runtimes: []string{"process"}, Tags: nil, CPU: 4, Memory: 4096, FreeCPU: 4, FreeMem: 4096, Kind: "linux"},
|
||||
{Hostname: "node-b", Runtimes: []string{"process"}, Tags: nil, CPU: 8, Memory: 8192, FreeCPU: 8, FreeMem: 8192, Kind: "linux"},
|
||||
{Hostname: "node-c", Runtimes: []string{"process"}, Tags: nil, CPU: 2, Memory: 2048, FreeCPU: 2, FreeMem: 2048, Kind: "linux"},
|
||||
}
|
||||
}
|
||||
|
||||
func jobSpec(name, oneOf string, constraints []string) *jobspec.WorkloadSpec {
|
||||
return &jobspec.WorkloadSpec{
|
||||
Kind: "Job",
|
||||
Name: name,
|
||||
Count: 1,
|
||||
Runtime: &jobspec.RuntimeBlock{OneOf: oneOf},
|
||||
Constraints: constraints,
|
||||
}
|
||||
}
|
||||
|
||||
func serviceSpec(name, oneOf string, count int, constraints []string) *jobspec.WorkloadSpec {
|
||||
return &jobspec.WorkloadSpec{
|
||||
Kind: "Service",
|
||||
Name: name,
|
||||
Count: count,
|
||||
Runtime: &jobspec.RuntimeBlock{OneOf: oneOf},
|
||||
Constraints: constraints,
|
||||
}
|
||||
}
|
||||
|
||||
func daemonSetSpec(name, oneOf string, constraints []string) *jobspec.WorkloadSpec {
|
||||
return &jobspec.WorkloadSpec{
|
||||
Kind: "DaemonSet",
|
||||
Name: name,
|
||||
Count: 1,
|
||||
Runtime: &jobspec.RuntimeBlock{OneOf: oneOf},
|
||||
Constraints: constraints,
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Job
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
func TestScheduleJob_BestFit(t *testing.T) {
|
||||
nodes := threeLinuxNodes()
|
||||
req := WorkloadRequest{Spec: jobSpec("batch", "process", nil), Namespace: "ns"}
|
||||
got, err := Schedule(nodes, req)
|
||||
if err != nil {
|
||||
t.Fatalf("Schedule: %v", err)
|
||||
}
|
||||
if len(got) != 1 {
|
||||
t.Fatalf("placements = %d, want 1", len(got))
|
||||
}
|
||||
if got[0].Node != "node-b" {
|
||||
t.Errorf("Node = %q, want node-b (most free capacity)", got[0].Node)
|
||||
}
|
||||
if !strings.HasPrefix(got[0].AllocID, "ns/batch-") {
|
||||
t.Errorf("AllocID = %q, want ns/batch-*", got[0].AllocID)
|
||||
}
|
||||
if got[0].Score <= 0 {
|
||||
t.Errorf("Score = %d, want > 0", got[0].Score)
|
||||
}
|
||||
}
|
||||
|
||||
func TestScheduleJob_NoFittingNode(t *testing.T) {
|
||||
nodes := threeLinuxNodes()
|
||||
// wasm runtime not advertised by any node.
|
||||
req := WorkloadRequest{Spec: jobSpec("wasmjob", "wasm", nil), Namespace: "ns"}
|
||||
if _, err := Schedule(nodes, req); err == nil {
|
||||
t.Fatal("Schedule: expected error for no-fitting node, got nil")
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Service
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
func TestScheduleService_SpreadAcrossNodes(t *testing.T) {
|
||||
nodes := threeLinuxNodes()
|
||||
req := WorkloadRequest{Spec: serviceSpec("web", "process", 3, nil), Namespace: "ns"}
|
||||
got, err := Schedule(nodes, req)
|
||||
if err != nil {
|
||||
t.Fatalf("Schedule: %v", err)
|
||||
}
|
||||
if len(got) != 3 {
|
||||
t.Fatalf("placements = %d, want 3", len(got))
|
||||
}
|
||||
seen := map[string]int{}
|
||||
for _, p := range got {
|
||||
seen[p.Node]++
|
||||
}
|
||||
if len(seen) != 3 {
|
||||
t.Errorf("anti-affinity spread: distinct nodes = %d, want 3; %v", len(seen), seen)
|
||||
}
|
||||
}
|
||||
|
||||
func TestScheduleService_ColocationWhenFewerNodes(t *testing.T) {
|
||||
nodes := threeLinuxNodes()
|
||||
req := WorkloadRequest{Spec: serviceSpec("web", "process", 5, nil), Namespace: "ns"}
|
||||
got, err := Schedule(nodes, req)
|
||||
if err != nil {
|
||||
t.Fatalf("Schedule: %v", err)
|
||||
}
|
||||
if len(got) != 5 {
|
||||
t.Fatalf("placements = %d, want 5", len(got))
|
||||
}
|
||||
seen := map[string]int{}
|
||||
for _, p := range got {
|
||||
seen[p.Node]++
|
||||
}
|
||||
if len(seen) != 3 {
|
||||
t.Errorf("colocation: distinct nodes = %d, want 3 (all used)", len(seen))
|
||||
}
|
||||
// No node should host more than 2 (3 nodes, 5 replicas: 2+2+1).
|
||||
for n, c := range seen {
|
||||
if c > 2 {
|
||||
t.Errorf("node %s has %d replicas, want <= 2", n, c)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestScheduleService_NoFittingNode(t *testing.T) {
|
||||
nodes := threeLinuxNodes()
|
||||
req := WorkloadRequest{Spec: serviceSpec("wasm-svc", "wasm", 3, nil), Namespace: "ns"}
|
||||
if _, err := Schedule(nodes, req); err == nil {
|
||||
t.Fatal("Schedule: expected error for service with no fitting node")
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// DaemonSet
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
func TestScheduleDaemonSet_AllMatching(t *testing.T) {
|
||||
nodes := threeLinuxNodes()
|
||||
req := WorkloadRequest{Spec: daemonSetSpec("logrotate", "process", nil), Namespace: "ns"}
|
||||
got, err := Schedule(nodes, req)
|
||||
if err != nil {
|
||||
t.Fatalf("Schedule: %v", err)
|
||||
}
|
||||
if len(got) != 3 {
|
||||
t.Errorf("placements = %d, want 3 (one per node)", len(got))
|
||||
}
|
||||
seen := map[string]bool{}
|
||||
for _, p := range got {
|
||||
seen[p.Node] = true
|
||||
}
|
||||
if len(seen) != 3 {
|
||||
t.Errorf("DaemonSet distinct nodes = %d, want 3", len(seen))
|
||||
}
|
||||
}
|
||||
|
||||
func TestScheduleDaemonSet_SomeExcludedByConstraint(t *testing.T) {
|
||||
nodes := threeLinuxNodes()
|
||||
// Only nodes with cpus >= 4 qualify: node-a (4) and node-b (8).
|
||||
req := WorkloadRequest{Spec: daemonSetSpec("heavy", "process", []string{"node.cpus >= 4"}), Namespace: "ns"}
|
||||
got, err := Schedule(nodes, req)
|
||||
if err != nil {
|
||||
t.Fatalf("Schedule: %v", err)
|
||||
}
|
||||
if len(got) != 2 {
|
||||
t.Errorf("placements = %d, want 2 (cpus>=4)", len(got))
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Runtime compatibility
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
func TestSchedule_RuntimeCompatibilityWasm(t *testing.T) {
|
||||
nodes := []NodeInfo{
|
||||
{Hostname: "no-wasm", Runtimes: []string{"process"}, Kind: "linux", CPU: 8, Memory: 8192, FreeCPU: 8, FreeMem: 8192},
|
||||
{Hostname: "has-wasm", Runtimes: []string{"process", "wasmtime"}, Kind: "linux", CPU: 4, Memory: 4096, FreeCPU: 4, FreeMem: 4096},
|
||||
}
|
||||
// Even though no-wasm has more free capacity, the wasm workload
|
||||
// must land on has-wasm.
|
||||
req := WorkloadRequest{Spec: jobSpec("wasmjob", "wasm", nil), Namespace: "ns"}
|
||||
got, err := Schedule(nodes, req)
|
||||
if err != nil {
|
||||
t.Fatalf("Schedule: %v", err)
|
||||
}
|
||||
if got[0].Node != "has-wasm" {
|
||||
t.Errorf("Node = %q, want has-wasm (runtime compatibility)", got[0].Node)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSchedule_RuntimeCompatibilityPveVM(t *testing.T) {
|
||||
nodes := []NodeInfo{
|
||||
{Hostname: "linux-1", Runtimes: []string{"process"}, Kind: "linux", CPU: 8, Memory: 8192, FreeCPU: 8, FreeMem: 8192},
|
||||
{Hostname: "pve-1", Runtimes: []string{"process"}, Kind: "proxmox", CPU: 8, Memory: 8192, FreeCPU: 8, FreeMem: 8192},
|
||||
}
|
||||
req := WorkloadRequest{Spec: jobSpec("vmjob", "pve-vm", nil), Namespace: "ns"}
|
||||
got, err := Schedule(nodes, req)
|
||||
if err != nil {
|
||||
t.Fatalf("Schedule: %v", err)
|
||||
}
|
||||
if got[0].Node != "pve-1" {
|
||||
t.Errorf("Node = %q, want pve-1 (pve-vm requires proxmox kind)", got[0].Node)
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Constraints
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
func TestSchedule_ConstraintKindExcludesProxmox(t *testing.T) {
|
||||
nodes := []NodeInfo{
|
||||
{Hostname: "linux-1", Runtimes: []string{"process"}, Kind: "linux", CPU: 8, Memory: 8192, FreeCPU: 8, FreeMem: 8192},
|
||||
{Hostname: "pve-1", Runtimes: []string{"process"}, Kind: "proxmox", CPU: 8, Memory: 8192, FreeCPU: 8, FreeMem: 8192},
|
||||
}
|
||||
req := WorkloadRequest{Spec: jobSpec("linuxonly", "process", []string{`node.kind == "linux"`}), Namespace: "ns"}
|
||||
got, err := Schedule(nodes, req)
|
||||
if err != nil {
|
||||
t.Fatalf("Schedule: %v", err)
|
||||
}
|
||||
if got[0].Node != "linux-1" {
|
||||
t.Errorf("Node = %q, want linux-1 (kind==linux)", got[0].Node)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSchedule_ConstraintCPUsExcludesSmall(t *testing.T) {
|
||||
nodes := threeLinuxNodes() // node-c has cpus=2
|
||||
req := WorkloadRequest{Spec: jobSpec("big", "process", []string{"node.cpus >= 4"}), Namespace: "ns"}
|
||||
got, err := Schedule(nodes, req)
|
||||
if err != nil {
|
||||
t.Fatalf("Schedule: %v", err)
|
||||
}
|
||||
if got[0].Node == "node-c" {
|
||||
t.Errorf("Node = node-c, want node-a or node-b (cpus>=4)")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSchedule_ConstraintNotInTags(t *testing.T) {
|
||||
nodes := []NodeInfo{
|
||||
{Hostname: "tagged", Runtimes: []string{"process"}, Tags: []string{"log-shipper"}, Kind: "linux", CPU: 8, Memory: 8192, FreeCPU: 8, FreeMem: 8192},
|
||||
{Hostname: "clean", Runtimes: []string{"process"}, Tags: nil, Kind: "linux", CPU: 4, Memory: 4096, FreeCPU: 4, FreeMem: 4096},
|
||||
}
|
||||
req := WorkloadRequest{Spec: jobSpec("worker", "process", []string{`"log-shipper" not in node.tags`}), Namespace: "ns"}
|
||||
got, err := Schedule(nodes, req)
|
||||
if err != nil {
|
||||
t.Fatalf("Schedule: %v", err)
|
||||
}
|
||||
if got[0].Node != "clean" {
|
||||
t.Errorf("Node = %q, want clean (log-shipper not in tags)", got[0].Node)
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Affinity
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
func TestSchedule_AffinityPrefersColocatedNode(t *testing.T) {
|
||||
// Place a redis service first, then a worker with affinity for
|
||||
// redis; the worker should prefer the node where redis already
|
||||
// runs even if another node has more free capacity.
|
||||
nodes := []NodeInfo{
|
||||
{Hostname: "big", Runtimes: []string{"process"}, Kind: "linux", CPU: 8, Memory: 8192, FreeCPU: 8, FreeMem: 8192},
|
||||
{Hostname: "small", Runtimes: []string{"process"}, Kind: "linux", CPU: 4, Memory: 4096, FreeCPU: 4, FreeMem: 4096},
|
||||
}
|
||||
redisReq := WorkloadRequest{Spec: serviceSpec("redis", "process", 1, nil), Namespace: "ns"}
|
||||
redisPlacements, err := Schedule(nodes, redisReq)
|
||||
if err != nil {
|
||||
t.Fatalf("redis Schedule: %v", err)
|
||||
}
|
||||
// Redis lands on "big" (most free capacity). Now schedule the
|
||||
// worker with affinity to redis; it should also land on "big".
|
||||
workerReq := WorkloadRequest{Spec: &jobspec.WorkloadSpec{
|
||||
Kind: "Job",
|
||||
Name: "worker",
|
||||
Count: 1,
|
||||
Runtime: &jobspec.RuntimeBlock{OneOf: "process"},
|
||||
Affinity: []jobspec.AffinityRule{
|
||||
{Target: "redis", Weight: 1000},
|
||||
},
|
||||
}, Namespace: "ns"}
|
||||
// The affinity is name-based; we need to seed the worker schedule
|
||||
// with the redis placement so affinityScore can see it. Schedule
|
||||
// does not take prior placements, so test affinityScore directly.
|
||||
got := affinityScore(nodes[0], workerReq, redisPlacements)
|
||||
if got <= 0 {
|
||||
t.Errorf("affinityScore(big) = %d, want > 0 (redis colocated)", got)
|
||||
}
|
||||
gotSmall := affinityScore(nodes[1], workerReq, redisPlacements)
|
||||
if gotSmall != 0 {
|
||||
t.Errorf("affinityScore(small) = %d, want 0 (redis not colocated)", gotSmall)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSchedule_AffinityCELExpression(t *testing.T) {
|
||||
// Affinity with a CEL target: prefer nodes tagged "ssd".
|
||||
nodes := []NodeInfo{
|
||||
{Hostname: "hdd", Runtimes: []string{"process"}, Tags: []string{"hdd"}, Kind: "linux", CPU: 8, Memory: 8192, FreeCPU: 8, FreeMem: 8192},
|
||||
{Hostname: "ssd", Runtimes: []string{"process"}, Tags: []string{"ssd"}, Kind: "linux", CPU: 4, Memory: 4096, FreeCPU: 4, FreeMem: 4096},
|
||||
}
|
||||
req := WorkloadRequest{Spec: &jobspec.WorkloadSpec{
|
||||
Kind: "Job",
|
||||
Name: "db",
|
||||
Count: 1,
|
||||
Runtime: &jobspec.RuntimeBlock{OneOf: "process"},
|
||||
Affinity: []jobspec.AffinityRule{
|
||||
{Target: `"ssd" in node.tags`, Weight: 10000},
|
||||
},
|
||||
}, Namespace: "ns"}
|
||||
got, err := Schedule(nodes, req)
|
||||
if err != nil {
|
||||
t.Fatalf("Schedule: %v", err)
|
||||
}
|
||||
if got[0].Node != "ssd" {
|
||||
t.Errorf("Node = %q, want ssd (affinity to ssd tag outweighs capacity)", got[0].Node)
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Error paths
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
func TestSchedule_EmptyNodes(t *testing.T) {
|
||||
req := WorkloadRequest{Spec: jobSpec("x", "process", nil), Namespace: "ns"}
|
||||
if _, err := Schedule(nil, req); err == nil {
|
||||
t.Fatal("Schedule: expected error for empty nodes, got nil")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSchedule_NilSpec(t *testing.T) {
|
||||
if _, err := Schedule(threeLinuxNodes(), WorkloadRequest{}); err == nil {
|
||||
t.Fatal("Schedule: expected error for nil spec, got nil")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSchedule_UnknownKind(t *testing.T) {
|
||||
req := WorkloadRequest{Spec: &jobspec.WorkloadSpec{Kind: "Cron", Name: "x", Count: 1}, Namespace: "ns"}
|
||||
if _, err := Schedule(threeLinuxNodes(), req); err == nil {
|
||||
t.Fatal("Schedule: expected error for unknown kind")
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Score unit tests
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
func TestScore_FitsAndDoesNotFit(t *testing.T) {
|
||||
node := NodeInfo{Hostname: "n", Runtimes: []string{"process"}, Kind: "linux", CPU: 4, Memory: 4096, FreeCPU: 4, FreeMem: 4096}
|
||||
req := WorkloadRequest{Spec: jobSpec("j", "process", nil), Namespace: "ns"}
|
||||
score, fits := Score(node, req)
|
||||
if !fits {
|
||||
t.Error("fits = false, want true")
|
||||
}
|
||||
if score <= 0 {
|
||||
t.Errorf("score = %d, want > 0", score)
|
||||
}
|
||||
}
|
||||
|
||||
func TestScore_RuntimeMismatchDoesNotFit(t *testing.T) {
|
||||
node := NodeInfo{Hostname: "n", Runtimes: []string{"process"}, Kind: "linux", CPU: 4, Memory: 4096, FreeCPU: 4, FreeMem: 4096}
|
||||
req := WorkloadRequest{Spec: jobSpec("j", "wasm", nil), Namespace: "ns"}
|
||||
if _, fits := Score(node, req); fits {
|
||||
t.Error("fits = true for wasm on process-only node, want false")
|
||||
}
|
||||
}
|
||||
|
||||
func TestScore_ConstraintFailsDoesNotFit(t *testing.T) {
|
||||
node := NodeInfo{Hostname: "n", Runtimes: []string{"process"}, Kind: "linux", CPU: 4, Memory: 4096, FreeCPU: 4, FreeMem: 4096}
|
||||
req := WorkloadRequest{Spec: jobSpec("j", "process", []string{`node.kind == "proxmox"`}), Namespace: "ns"}
|
||||
if _, fits := Score(node, req); fits {
|
||||
t.Error("fits = true for kind==proxmox on linux node, want false")
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// allocID / isAllocFor helpers
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
func TestAllocID(t *testing.T) {
|
||||
req := WorkloadRequest{Spec: &jobspec.WorkloadSpec{Name: "web"}, Namespace: "prod"}
|
||||
if got := allocID(req, 2); got != "prod/web-2" {
|
||||
t.Errorf("allocID = %q, want prod/web-2", got)
|
||||
}
|
||||
req.Namespace = ""
|
||||
if got := allocID(req, 0); got != "default/web-0" {
|
||||
t.Errorf("allocID = %q, want default/web-0", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestIsAllocFor(t *testing.T) {
|
||||
cases := []struct {
|
||||
allocID string
|
||||
workload string
|
||||
want bool
|
||||
}{
|
||||
{"ns/redis-0", "redis", true},
|
||||
{"ns/redis-12", "redis", true},
|
||||
{"ns/worker-0", "redis", false},
|
||||
{"redis-0", "redis", true},
|
||||
{"ns/web-canary-3", "web-canary", true},
|
||||
}
|
||||
for _, c := range cases {
|
||||
if got := isAllocFor(c.allocID, c.workload); got != c.want {
|
||||
t.Errorf("isAllocFor(%q,%q) = %v, want %v", c.allocID, c.workload, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// normalizeRuntime / hasRuntime
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
func TestHasRuntimeAliases(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
node NodeInfo
|
||||
want bool
|
||||
}{
|
||||
{"wasm on wasmtime node", NodeInfo{Runtimes: []string{"wasmtime"}, Kind: "linux"}, true},
|
||||
{"wasm on process node", NodeInfo{Runtimes: []string{"process"}, Kind: "linux"}, false},
|
||||
{"pve-vm on linux node", NodeInfo{Runtimes: []string{"pve-vm"}, Kind: "linux"}, false},
|
||||
{"pve-vm on proxmox node", NodeInfo{Runtimes: nil, Kind: "proxmox"}, true},
|
||||
{"process on process node", NodeInfo{Runtimes: []string{"process"}, Kind: "linux"}, true},
|
||||
{"empty runtime on any node", NodeInfo{Runtimes: []string{"process"}, Kind: "linux"}, true},
|
||||
}
|
||||
for _, c := range cases {
|
||||
if c.name == "empty runtime on any node" {
|
||||
// hasRuntime is only called when OneOf != "".
|
||||
continue
|
||||
}
|
||||
if got := hasRuntime(c.node, "wasm"); c.name == "wasm on wasmtime node" || c.name == "wasm on process node" {
|
||||
if got != c.want {
|
||||
t.Errorf("%s: hasRuntime(wasm) = %v, want %v", c.name, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
// Explicit pve-vm and process checks.
|
||||
if !hasRuntime(NodeInfo{Runtimes: nil, Kind: "proxmox"}, "pve-vm") {
|
||||
t.Error("pve-vm on proxmox node should fit")
|
||||
}
|
||||
if hasRuntime(NodeInfo{Runtimes: nil, Kind: "linux"}, "pve-vm") {
|
||||
t.Error("pve-vm on linux node should not fit")
|
||||
}
|
||||
if !hasRuntime(NodeInfo{Runtimes: []string{"process"}, Kind: "linux"}, "process") {
|
||||
t.Error("process on process node should fit")
|
||||
}
|
||||
}
|
||||
@@ -37,6 +37,8 @@ type Validator interface {
|
||||
// - count must be 1 (or unset → 1); count > 1 is an error for Job
|
||||
// (use a Service for replicas)
|
||||
// - no Traefik route (a ServiceBlock is rejected)
|
||||
// - task group (spec.Tasks) optional; when present, each task must
|
||||
// have a unique name and a resolvable command (P06).
|
||||
type JobValidator struct{}
|
||||
|
||||
// ServiceValidator validates the Service workload kind (R-012).
|
||||
@@ -46,11 +48,14 @@ type JobValidator struct{}
|
||||
// - count ≥ 1
|
||||
// - restart required (mode must be service)
|
||||
// - update required (strategy must be rolling/canary/blue-green)
|
||||
// - runtime required
|
||||
// - runtime required (unless a task group is present; each task
|
||||
// can carry its own runtime — P06)
|
||||
// - health block required (Traefik routing depends on health checks)
|
||||
// - service block, if present, must have a valid bind (127.0.0.1
|
||||
// opt-in per R-007; default is socket — empty bind is OK)
|
||||
// - service block implied (Traefik route YES)
|
||||
// - task group (spec.Tasks) optional; when present, each task must
|
||||
// have a unique name and a resolvable command (P06).
|
||||
type ServiceValidator struct{}
|
||||
|
||||
// DaemonSetValidator validates the DaemonSet workload kind (R-012).
|
||||
@@ -60,6 +65,8 @@ type ServiceValidator struct{}
|
||||
// - no ports (no Traefik route by default D-175)
|
||||
// - no count (implicit = nodes matching condition)
|
||||
// - restart required
|
||||
// - task group (spec.Tasks) optional; when present, each task must
|
||||
// have a unique name and a resolvable command (P06).
|
||||
type DaemonSetValidator struct{}
|
||||
|
||||
// ValidatorFor returns the Validator for the given workload kind, or an
|
||||
@@ -93,6 +100,7 @@ func (JobValidator) Validate(spec *jobspec.WorkloadSpec) error {
|
||||
if spec.Service != nil {
|
||||
errs = append(errs, "service block (Traefik route) is not allowed for Job (D-175)")
|
||||
}
|
||||
errs = append(errs, validateTaskGroup(spec)...)
|
||||
return composeErrors("schema/Job", errs)
|
||||
}
|
||||
|
||||
@@ -137,8 +145,8 @@ func (ServiceValidator) Validate(spec *jobspec.WorkloadSpec) error {
|
||||
errs = append(errs, fmt.Sprintf("update strategy %q invalid (want one of rolling, canary, blue-green)", spec.Update.Strategy))
|
||||
}
|
||||
}
|
||||
if spec.Runtime == nil {
|
||||
errs = append(errs, "runtime block required for Service")
|
||||
if spec.Runtime == nil && len(spec.Tasks) == 0 {
|
||||
errs = append(errs, "runtime block required for Service (or a task group with per-task runtimes)")
|
||||
}
|
||||
if spec.Health == nil {
|
||||
errs = append(errs, "health block required for Service (Traefik routing requires health checks)")
|
||||
@@ -148,6 +156,7 @@ func (ServiceValidator) Validate(spec *jobspec.WorkloadSpec) error {
|
||||
errs = append(errs, err.Error())
|
||||
}
|
||||
}
|
||||
errs = append(errs, validateTaskGroup(spec)...)
|
||||
return composeErrors("schema/Service", errs)
|
||||
}
|
||||
|
||||
@@ -195,6 +204,7 @@ func (DaemonSetValidator) Validate(spec *jobspec.WorkloadSpec) error {
|
||||
if spec.Restart == nil {
|
||||
errs = append(errs, "restart block required for DaemonSet")
|
||||
}
|
||||
errs = append(errs, validateTaskGroup(spec)...)
|
||||
return composeErrors("schema/DaemonSet", errs)
|
||||
}
|
||||
|
||||
@@ -207,3 +217,38 @@ func composeErrors(name string, errs []string) error {
|
||||
}
|
||||
return fmt.Errorf("%s: %s", name, strings.Join(errs, "; "))
|
||||
}
|
||||
|
||||
// validateTaskGroup validates the task-group list shared by all kinds
|
||||
// (P06, PRD §9.1). When the spec carries a task group (spec.Tasks
|
||||
// non-empty), each task must have a unique name and a resolvable
|
||||
// command (the task's own Command, the task's runtime command, or the
|
||||
// top-level runtime command as the per-group default). The top-level
|
||||
// runtime is optional when tasks is present (each task can carry its
|
||||
// own runtime). Returns nil when the spec has no task group.
|
||||
func validateTaskGroup(spec *jobspec.WorkloadSpec) []string {
|
||||
if len(spec.Tasks) == 0 {
|
||||
return nil
|
||||
}
|
||||
var errs []string
|
||||
seen := make(map[string]bool, len(spec.Tasks))
|
||||
for i, task := range spec.Tasks {
|
||||
if strings.TrimSpace(task.Name) == "" {
|
||||
errs = append(errs, fmt.Sprintf("tasks[%d]: name is required", i))
|
||||
} else if seen[task.Name] {
|
||||
errs = append(errs, fmt.Sprintf("tasks[%d]: duplicate task name %q (names must be unique within the group)", i, task.Name))
|
||||
} else {
|
||||
seen[task.Name] = true
|
||||
}
|
||||
cmd := task.Command
|
||||
if strings.TrimSpace(cmd) == "" && task.Runtime != nil {
|
||||
cmd = task.Runtime.Command
|
||||
}
|
||||
if strings.TrimSpace(cmd) == "" && spec.Runtime != nil {
|
||||
cmd = spec.Runtime.Command
|
||||
}
|
||||
if strings.TrimSpace(cmd) == "" {
|
||||
errs = append(errs, fmt.Sprintf("tasks[%d]: command is required (set tasks[].command, tasks[].runtime.command, or top-level runtime.command)", i))
|
||||
}
|
||||
}
|
||||
return errs
|
||||
}
|
||||
|
||||
@@ -689,3 +689,183 @@ var (
|
||||
_ Validator = ServiceValidator{}
|
||||
_ Validator = DaemonSetValidator{}
|
||||
)
|
||||
|
||||
func TestTaskGroup_Valid(t *testing.T) {
|
||||
// P06: a valid task group — two tasks, each with a unique name
|
||||
// and a resolvable command (own command). The top-level runtime
|
||||
// is optional when each task carries its own.
|
||||
spec := &jobspec.WorkloadSpec{
|
||||
Kind: "Service",
|
||||
Name: "web",
|
||||
Count: 1,
|
||||
Tasks: []jobspec.TaskGroupTask{
|
||||
{Name: "app", Command: "/usr/bin/httpd"},
|
||||
{Name: "sidecar", Command: "/bin/wasm-runner sidecar.wasm"},
|
||||
},
|
||||
Restart: &jobspec.RestartBlock{Mode: "service"},
|
||||
Update: &jobspec.UpdateBlock{Strategy: "rolling"},
|
||||
Health: &jobspec.HealthBlock{CheckType: "http"},
|
||||
Ports: []jobspec.PortSpec{{Name: "http", Port: 8080}},
|
||||
}
|
||||
if err := (ServiceValidator{}).Validate(spec); err != nil {
|
||||
t.Fatalf("expected nil, got %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestTaskGroup_ValidInheritsTopLevelRuntime(t *testing.T) {
|
||||
// P06: tasks without their own runtime inherit the top-level
|
||||
// runtime command. The validator accepts this as long as the
|
||||
// resolved command is non-empty.
|
||||
spec := &jobspec.WorkloadSpec{
|
||||
Kind: "Service",
|
||||
Name: "web",
|
||||
Count: 1,
|
||||
Runtime: &jobspec.RuntimeBlock{OneOf: "process", Command: "/bin/default"},
|
||||
Tasks: []jobspec.TaskGroupTask{
|
||||
{Name: "app"},
|
||||
{Name: "sidecar"},
|
||||
},
|
||||
Restart: &jobspec.RestartBlock{Mode: "service"},
|
||||
Update: &jobspec.UpdateBlock{Strategy: "rolling"},
|
||||
Health: &jobspec.HealthBlock{CheckType: "http"},
|
||||
Ports: []jobspec.PortSpec{{Name: "http", Port: 8080}},
|
||||
}
|
||||
if err := (ServiceValidator{}).Validate(spec); err != nil {
|
||||
t.Fatalf("expected nil, got %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestTaskGroup_ValidTaskRuntimeCommand(t *testing.T) {
|
||||
// P06: a task whose command is provided via the task's own
|
||||
// runtime.command (no top-level runtime) is valid.
|
||||
spec := &jobspec.WorkloadSpec{
|
||||
Kind: "Service",
|
||||
Name: "web",
|
||||
Count: 1,
|
||||
Tasks: []jobspec.TaskGroupTask{
|
||||
{Name: "app", Runtime: &jobspec.RuntimeBlock{OneOf: "process", Command: "/usr/bin/httpd"}},
|
||||
},
|
||||
Restart: &jobspec.RestartBlock{Mode: "service"},
|
||||
Update: &jobspec.UpdateBlock{Strategy: "rolling"},
|
||||
Health: &jobspec.HealthBlock{CheckType: "http"},
|
||||
Ports: []jobspec.PortSpec{{Name: "http", Port: 8080}},
|
||||
}
|
||||
if err := (ServiceValidator{}).Validate(spec); err != nil {
|
||||
t.Fatalf("expected nil, got %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestTaskGroup_MissingTaskName(t *testing.T) {
|
||||
spec := &jobspec.WorkloadSpec{
|
||||
Kind: "Service",
|
||||
Name: "web",
|
||||
Tasks: []jobspec.TaskGroupTask{
|
||||
{Command: "/usr/bin/httpd"},
|
||||
{Name: "sidecar", Command: "/bin/wasm-runner"},
|
||||
},
|
||||
Restart: &jobspec.RestartBlock{Mode: "service"},
|
||||
Update: &jobspec.UpdateBlock{Strategy: "rolling"},
|
||||
Health: &jobspec.HealthBlock{CheckType: "http"},
|
||||
Ports: []jobspec.PortSpec{{Name: "http", Port: 8080}},
|
||||
}
|
||||
err := ServiceValidator{}.Validate(spec)
|
||||
if err == nil {
|
||||
t.Fatal("expected error for missing task name, got nil")
|
||||
}
|
||||
if !strings.Contains(err.Error(), "name is required") {
|
||||
t.Errorf("error = %q, want 'name is required'", err.Error())
|
||||
}
|
||||
}
|
||||
|
||||
func TestTaskGroup_DuplicateTaskNames(t *testing.T) {
|
||||
spec := &jobspec.WorkloadSpec{
|
||||
Kind: "Service",
|
||||
Name: "web",
|
||||
Tasks: []jobspec.TaskGroupTask{
|
||||
{Name: "app", Command: "/usr/bin/httpd"},
|
||||
{Name: "app", Command: "/bin/other"},
|
||||
},
|
||||
Restart: &jobspec.RestartBlock{Mode: "service"},
|
||||
Update: &jobspec.UpdateBlock{Strategy: "rolling"},
|
||||
Health: &jobspec.HealthBlock{CheckType: "http"},
|
||||
Ports: []jobspec.PortSpec{{Name: "http", Port: 8080}},
|
||||
}
|
||||
err := ServiceValidator{}.Validate(spec)
|
||||
if err == nil {
|
||||
t.Fatal("expected error for duplicate task names, got nil")
|
||||
}
|
||||
if !strings.Contains(err.Error(), "duplicate task name") {
|
||||
t.Errorf("error = %q, want 'duplicate task name'", err.Error())
|
||||
}
|
||||
}
|
||||
|
||||
func TestTaskGroup_MissingCommand(t *testing.T) {
|
||||
// P06: a task with no resolvable command (no task.Command, no
|
||||
// task.Runtime, no top-level Runtime) is rejected.
|
||||
spec := &jobspec.WorkloadSpec{
|
||||
Kind: "Service",
|
||||
Name: "web",
|
||||
Tasks: []jobspec.TaskGroupTask{
|
||||
{Name: "app"},
|
||||
},
|
||||
Restart: &jobspec.RestartBlock{Mode: "service"},
|
||||
Update: &jobspec.UpdateBlock{Strategy: "rolling"},
|
||||
Health: &jobspec.HealthBlock{CheckType: "http"},
|
||||
Ports: []jobspec.PortSpec{{Name: "http", Port: 8080}},
|
||||
}
|
||||
err := ServiceValidator{}.Validate(spec)
|
||||
if err == nil {
|
||||
t.Fatal("expected error for missing task command, got nil")
|
||||
}
|
||||
if !strings.Contains(err.Error(), "command is required") {
|
||||
t.Errorf("error = %q, want 'command is required'", err.Error())
|
||||
}
|
||||
}
|
||||
|
||||
func TestTaskGroup_JobAcceptsTaskGroup(t *testing.T) {
|
||||
// P06: task groups apply to all kinds, not just Service. Job
|
||||
// accepts a task group with unique names + resolvable commands.
|
||||
spec := &jobspec.WorkloadSpec{
|
||||
Kind: "Job",
|
||||
Name: "batch",
|
||||
Count: 1,
|
||||
Tasks: []jobspec.TaskGroupTask{
|
||||
{Name: "step1", Command: "/bin/extract"},
|
||||
{Name: "step2", Command: "/bin/transform"},
|
||||
},
|
||||
}
|
||||
if err := (JobValidator{}).Validate(spec); err != nil {
|
||||
t.Fatalf("expected nil, got %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestTaskGroup_DaemonSetAcceptsTaskGroup(t *testing.T) {
|
||||
// P06: DaemonSet accepts a task group.
|
||||
spec := &jobspec.WorkloadSpec{
|
||||
Kind: "DaemonSet",
|
||||
Name: "log-shipper",
|
||||
Schedule: &jobspec.ScheduleBlock{Mode: "every-node"},
|
||||
Restart: &jobspec.RestartBlock{Mode: "on-failure"},
|
||||
Tasks: []jobspec.TaskGroupTask{
|
||||
{Name: "collector", Command: "/bin/collect"},
|
||||
{Name: "forwarder", Command: "/bin/forward"},
|
||||
},
|
||||
}
|
||||
if err := (DaemonSetValidator{}).Validate(spec); err != nil {
|
||||
t.Fatalf("expected nil, got %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestTaskGroup_NoTasksBackwardCompat(t *testing.T) {
|
||||
// Backward compat: a spec with no Tasks is validated by the
|
||||
// existing kind-specific rules (no task-group check fires).
|
||||
spec := &jobspec.WorkloadSpec{
|
||||
Kind: "Job",
|
||||
Name: "backup",
|
||||
Count: 1,
|
||||
Runtime: &jobspec.RuntimeBlock{OneOf: "process", Command: "/bin/rsync"},
|
||||
}
|
||||
if err := (JobValidator{}).Validate(spec); err != nil {
|
||||
t.Fatalf("expected nil, got %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user