fix(P07): concurrency safety — SQLite, flock, cache, atomic writes (REQ-156)
- SQLite busy_timeout(5000) + SetMaxOpenConns(1) on all 4 DSNs - secrets file flock (concurrent set on same ns no longer loses data) - upgrade lock file (refuse concurrent orca upgrade) - backup lock file (refuse concurrent backup) - cache invalidation by writes (read-after-write consistency) - Executor.Run mutex scope fix (hold only for DB inserts) - ns create/inherit/set-constraint atomic writeNSMdAtomic - writeCurrentLead + rotateSSHKeys atomic - consolidate 3 writeAtomic impls onto security.WriteAtomic - WebAuthn session stores guarded with sync.Mutex Tests: concurrent secrets set, upgrade lock rejection, cache read-after-write, WebAuthn session thread-safety (pass under -race). ---ci--- project: orca phase: 7 milestone: v0.13 status: complete requirements: covered: [156] ---/ci---
This commit is contained in:
@@ -98,24 +98,39 @@ type TaskSpec struct {
|
||||
}
|
||||
|
||||
func (e *Executor) Run(ctx context.Context, job *model.Job, specs []TaskSpec) error {
|
||||
e.mu.Lock()
|
||||
defer e.mu.Unlock()
|
||||
// REQ-156 / P07 T6: the mutex previously guarded the ENTIRE job
|
||||
// (insert + status transitions + task execution + wait). That
|
||||
// serialized unrelated jobs against each other and held the lock
|
||||
// across long-running child processes, blocking concurrent
|
||||
// Submit/Status/Run callers. The mutex is now scoped ONLY to the
|
||||
// DB inserts/updates (the part that must be serialized against
|
||||
// the single-writer SQLite connection pool — see store.Open
|
||||
// SetMaxOpenConns(1)). The task goroutines spawned below do not
|
||||
// hold e.mu; they share the per-job failure counter via a local
|
||||
// sync.Mutex.
|
||||
|
||||
// Insert the job first so tasks can reference it via foreign key.
|
||||
// Insert the job + flip to Running under the lock (serializes
|
||||
// the DB writes; the underlying SQLite busy_timeout(5000) +
|
||||
// SetMaxOpenConns(1) handles contention).
|
||||
e.mu.Lock()
|
||||
if err := e.jobs.Insert(ctx, job); err != nil {
|
||||
e.mu.Unlock()
|
||||
return err
|
||||
}
|
||||
if err := e.jobs.UpdateStatus(ctx, job.ID, model.JobStatusRunning, 0); err != nil {
|
||||
e.mu.Unlock()
|
||||
return err
|
||||
}
|
||||
e.mu.Unlock()
|
||||
|
||||
// Task execution runs WITHOUT e.mu — concurrent jobs (and
|
||||
// concurrent Submit/Status callers) are no longer blocked by a
|
||||
// long-running child process.
|
||||
var (
|
||||
wg sync.WaitGroup
|
||||
failedCount int
|
||||
exitCode int
|
||||
mu sync.Mutex
|
||||
)
|
||||
|
||||
for _, ts := range specs {
|
||||
wg.Add(1)
|
||||
go func(ts TaskSpec) {
|
||||
@@ -133,14 +148,16 @@ func (e *Executor) Run(ctx context.Context, job *model.Job, specs []TaskSpec) er
|
||||
}
|
||||
wg.Wait()
|
||||
|
||||
// Final status transition under the lock (the DB write is the
|
||||
// only thing that needs serialization).
|
||||
e.mu.Lock()
|
||||
defer e.mu.Unlock()
|
||||
if failedCount > 0 {
|
||||
exitCode = 1
|
||||
if err := e.jobs.UpdateStatus(ctx, job.ID, model.JobStatusFailed, exitCode); err != nil {
|
||||
if err := e.jobs.UpdateStatus(ctx, job.ID, model.JobStatusFailed, 1); err != nil {
|
||||
return err
|
||||
}
|
||||
return fmt.Errorf("%d/%d tasks failed", failedCount, len(specs))
|
||||
}
|
||||
|
||||
if err := e.jobs.UpdateStatus(ctx, job.ID, model.JobStatusComplete, 0); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user