13493: Merge branch 'master' into 13493-federation-proxy

[arvados.git] / services / crunch-dispatch-slurm / squeue.go
diff --git a/services/crunch-dispatch-slurm/squeue.go b/services/crunch-dispatch-slurm/squeue.go

index b8e3108c7c1a235e7c7e28ffc10974652a35cc6a..fd4851eb0a8a92b48fcacef0e4552ce99d0a7f48 100644 (file)
--- a/services/crunch-dispatch-slurm/squeue.go
+++ b/services/crunch-dispatch-slurm/squeue.go
@@ -14,11 +14,14 @@ import (
         "time"
  )
  
+const slurm15NiceLimit int64 = 10000
+
  type slurmJob struct {
         uuid         string
         wantPriority int64
         priority     int64 // current slurm priority (incorporates nice value)
         nice         int64 // current slurm nice value
+       hitNiceLimit bool
  }
  
  // Squeue implements asynchronous polling monitor of the SLURM queue using the
@@ -79,19 +82,42 @@ func (sqc *SqueueChecker) reniceAll() {
                         // (perhaps it's not an Arvados job)
                         continue
                 }
+               if j.priority == 0 {
+                       // SLURM <= 15.x implements "hold" by setting
+                       // priority to 0. If we include held jobs
+                       // here, we'll end up trying to push other
+                       // jobs below them using negative priority,
+                       // which won't help anything.
+                       continue
+               }
                 jobs = append(jobs, j)
         }
  
         sort.Slice(jobs, func(i, j int) bool {
-               return jobs[i].wantPriority > jobs[j].wantPriority
+               if jobs[i].wantPriority != jobs[j].wantPriority {
+                       return jobs[i].wantPriority > jobs[j].wantPriority
+               } else {
+                       // break ties with container uuid --
+                       // otherwise, the ordering would change from
+                       // one interval to the next, and we'd do many
+                       // pointless slurm queue rearrangements.
+                       return jobs[i].uuid > jobs[j].uuid
+               }
         })
         renice := wantNice(jobs, sqc.PrioritySpread)
         for i, job := range jobs {
-               if renice[i] == job.nice {
+               niceNew := renice[i]
+               if job.hitNiceLimit && niceNew > slurm15NiceLimit {
+                       niceNew = slurm15NiceLimit
+               }
+               if niceNew == job.nice {
                         continue
                 }
-               log.Printf("updating slurm priority for %q: nice %d => %d", job.uuid, job.nice, renice[i])
-               sqc.Slurm.Renice(job.uuid, renice[i])
+               err := sqc.Slurm.Renice(job.uuid, niceNew)
+               if err != nil && niceNew > slurm15NiceLimit && strings.Contains(err.Error(), "Invalid nice value") {
+                       log.Printf("container %q clamping nice values at %d, priority order will not be correct -- see https://dev.arvados.org/projects/arvados/wiki/SLURM_integration#Limited-nice-values-SLURM-15", job.uuid, slurm15NiceLimit)
+                       job.hitNiceLimit = true
+               }
         }
  }
  
@@ -114,7 +140,7 @@ func (sqc *SqueueChecker) check() {
         sqc.L.Lock()
         defer sqc.L.Unlock()
  
-       cmd := sqc.Slurm.QueueCommand([]string{"--all", "--format=%j %y %Q"})
+       cmd := sqc.Slurm.QueueCommand([]string{"--all", "--noheader", "--format=%j %y %Q %T %r"})
         stdout, stderr := &bytes.Buffer{}, &bytes.Buffer{}
         cmd.Stdout, cmd.Stderr = stdout, stderr
         if err := cmd.Run(); err != nil {
@@ -128,9 +154,9 @@ func (sqc *SqueueChecker) check() {
                 if line == "" {
                         continue
                 }
-               var uuid string
+               var uuid, state, reason string
                 var n, p int64
-               if _, err := fmt.Sscan(line, &uuid, &n, &p); err != nil {
+               if _, err := fmt.Sscan(line, &uuid, &n, &p, &state, &reason); err != nil {
                         log.Printf("warning: ignoring unparsed line in squeue output: %q", line)
                         continue
                 }
@@ -141,6 +167,33 @@ func (sqc *SqueueChecker) check() {
                 replacing.priority = p
                 replacing.nice = n
                 newq[uuid] = replacing
+
+               if state == "PENDING" && ((reason == "BadConstraints" && p <= 2*slurm15NiceLimit) || reason == "launch failed requeued held") && replacing.wantPriority > 0 {
+                       // When using SLURM 14.x or 15.x, our queued
+                       // jobs land in this state when "scontrol
+                       // reconfigure" invalidates their feature
+                       // constraints by clearing all node features.
+                       // They stay in this state even after the
+                       // features reappear, until we run "scontrol
+                       // release {jobid}". Priority is usually 0 in
+                       // this state, but sometimes (due to a race
+                       // with nice adjustments?) it's a small
+                       // positive value.
+                       //
+                       // "scontrol release" is silent and successful
+                       // regardless of whether the features have
+                       // reappeared, so rather than second-guessing
+                       // whether SLURM is ready, we just keep trying
+                       // this until it works.
+                       //
+                       // "launch failed requeued held" seems to be
+                       // another manifestation of this problem,
+                       // resolved the same way.
+                       log.Printf("releasing held job %q (priority=%d, state=%q, reason=%q)", uuid, p, state, reason)
+                       sqc.Slurm.Release(uuid)
+               } else if p < 1<<20 && replacing.wantPriority > 0 {
+                       log.Printf("warning: job %q has low priority %d, nice %d, state %q, reason %q", uuid, p, n, state, reason)
+               }
         }
         sqc.queue = newq
         sqc.Broadcast()