X-Git-Url: https://git.arvados.org/arvados.git/blobdiff_plain/19ae770973482257117fe8ded5619c3018c4b60f..a8378b8deaa2bbf9d2c154d9d9bb072538c288cc:/services/nodemanager/arvnodeman/jobqueue.py diff --git a/services/nodemanager/arvnodeman/jobqueue.py b/services/nodemanager/arvnodeman/jobqueue.py index 87cf738311..f6e9249ebb 100644 --- a/services/nodemanager/arvnodeman/jobqueue.py +++ b/services/nodemanager/arvnodeman/jobqueue.py @@ -3,6 +3,7 @@ from __future__ import absolute_import, print_function import logging +import subprocess from . import clientactor from .config import ARVADOS_ERRORS @@ -18,13 +19,14 @@ class ServerCalculator(object): """ class CloudSizeWrapper(object): - def __init__(self, real_size, **kwargs): + def __init__(self, real_size, node_mem_scaling, **kwargs): self.real = real_size for name in ['id', 'name', 'ram', 'disk', 'bandwidth', 'price', 'extra']: setattr(self, name, getattr(self.real, name)) self.cores = kwargs.pop('cores') self.scratch = self.disk + self.ram = int(self.ram * node_mem_scaling) for name, override in kwargs.iteritems(): if not hasattr(self, name): raise ValueError("unrecognized size field '%s'" % (name,)) @@ -41,8 +43,9 @@ class ServerCalculator(object): return True - def __init__(self, server_list, max_nodes=None, max_price=None): - self.cloud_sizes = [self.CloudSizeWrapper(s, **kws) + def __init__(self, server_list, max_nodes=None, max_price=None, + node_mem_scaling=0.95): + self.cloud_sizes = [self.CloudSizeWrapper(s, node_mem_scaling, **kws) for s, kws in server_list] self.cloud_sizes.sort(key=lambda s: s.price) self.max_nodes = max_nodes or float('inf') @@ -109,7 +112,24 @@ class JobQueueMonitorActor(clientactor.RemotePollLoopActor): self._calculator = server_calc def _send_request(self): - return self._client.jobs().queue().execute()['items'] + # cpus, memory, tempory disk space, reason, job name + squeue_out = subprocess.check_output(["squeue", "--state=PENDING", "--noheader", "--format=%c %m %d %r %j"]) + queuelist = [] + for out in squeue_out.splitlines(): + cpu, ram, disk, reason, jobname = out.split(" ", 4) + if reason in ("Resources", "ReqNodeNotAvail"): + queuelist.append({ + "uuid": jobname, + "runtime_constraints": { + "min_cores_per_node": cpu, + "min_ram_mb_per_node": ram, + "min_scratch_mb_per_node": disk + } + }) + + queuelist.extend(self._client.jobs().queue().execute()['items']) + + return queuelist def _got_response(self, queue): server_list = self._calculator.servers_for_queue(queue)