13306: Changes to arvados-cwl-runner code after running futurize --stage2

[arvados.git] / sdk / cwl / arvados_cwl / arvcontainer.py
diff --git a/sdk/cwl/arvados_cwl/arvcontainer.py b/sdk/cwl/arvados_cwl/arvcontainer.py

index 0bec692643ad805c02d6b8358fae8a65841c1367..d2f04b5dd0f7b1df07f06f6d20eeeb5341945722 100644 (file)
--- a/sdk/cwl/arvados_cwl/arvcontainer.py
+++ b/sdk/cwl/arvados_cwl/arvcontainer.py
@@ -1,3 +1,6 @@
+from future import standard_library
+standard_library.install_aliases()
+from builtins import str
  # Copyright (C) The Arvados Authors. All rights reserved.
  #
  # SPDX-License-Identifier: Apache-2.0
@@ -5,18 +8,21 @@
  import logging
  import json
  import os
-import urllib
+import urllib.request, urllib.parse, urllib.error
  import time
  import datetime
  import ciso8601
  import uuid
+import math
  
+import arvados_cwl.util
  import ruamel.yaml as yaml
  
  from cwltool.errors import WorkflowException
-from cwltool.process import get_feature, UnsupportedRequirement, shortname
+from cwltool.process import UnsupportedRequirement, shortname
  from cwltool.pathmapper import adjustFileObjs, adjustDirObjs, visit_class
  from cwltool.utils import aslist
+from cwltool.job import JobBase
  
  import arvados.collection
  
@@ -30,18 +36,27 @@ from .perf import Perf
  logger = logging.getLogger('arvados.cwl-runner')
  metrics = logging.getLogger('arvados.cwl-runner.metrics')
  
-class ArvadosContainer(object):
+class ArvadosContainer(JobBase):
      """Submit and manage a Crunch container request for executing a CWL CommandLineTool."""
  
-    def __init__(self, runner):
+    def __init__(self, runner, job_runtime,
+                 builder,   # type: Builder
+                 joborder,  # type: Dict[Text, Union[Dict[Text, Any], List, Text]]
+                 make_path_mapper,  # type: Callable[..., PathMapper]
+                 requirements,      # type: List[Dict[Text, Text]]
+                 hints,     # type: List[Dict[Text, Text]]
+                 name       # type: Text
+    ):
+        super(ArvadosContainer, self).__init__(builder, joborder, make_path_mapper, requirements, hints, name)
          self.arvrunner = runner
+        self.job_runtime = job_runtime
          self.running = False
          self.uuid = None
  
      def update_pipeline_component(self, r):
          pass
  
-    def run(self, dry_run=False, pull_image=True, **kwargs):
+    def run(self, runtimeContext):
          # ArvadosCommandTool subclasses from cwltool.CommandLineTool,
          # which calls makeJobRunner() to get a new ArvadosContainer
          # object.  The fields that define execution such as
@@ -49,19 +64,21 @@ class ArvadosContainer(object):
          # ArvadosContainer object by CommandLineTool.job() before
          # run() is called.
  
+        runtimeContext = self.job_runtime
+
          container_request = {
              "command": self.command_line,
              "name": self.name,
              "output_path": self.outdir,
              "cwd": self.outdir,
-            "priority": kwargs.get("priority"),
+            "priority": runtimeContext.priority,
              "state": "Committed",
              "properties": {},
          }
          runtime_constraints = {}
  
-        if self.arvrunner.project_uuid:
-            container_request["owner_uuid"] = self.arvrunner.project_uuid
+        if runtimeContext.project_uuid:
+            container_request["owner_uuid"] = runtimeContext.project_uuid
  
          if self.arvrunner.secret_store.has_secret(self.command_line):
              raise WorkflowException("Secret material leaked on command line, only file literals may contain secrets")
@@ -71,17 +88,17 @@ class ArvadosContainer(object):
  
          resources = self.builder.resources
          if resources is not None:
-            runtime_constraints["vcpus"] = resources.get("cores", 1)
-            runtime_constraints["ram"] = resources.get("ram") * 2**20
+            runtime_constraints["vcpus"] = math.ceil(resources.get("cores", 1))
+            runtime_constraints["ram"] = math.ceil(resources.get("ram") * 2**20)
  
          mounts = {
              self.outdir: {
                  "kind": "tmp",
-                "capacity": resources.get("outdirSize", 0) * 2**20
+                "capacity": math.ceil(resources.get("outdirSize", 0) * 2**20)
              },
              self.tmpdir: {
                  "kind": "tmp",
-                "capacity": resources.get("tmpdirSize", 0) * 2**20
+                "capacity": math.ceil(resources.get("tmpdirSize", 0) * 2**20)
              }
          }
          secret_mounts = {}
@@ -119,10 +136,10 @@ class ArvadosContainer(object):
                  vwd = arvados.collection.Collection(api_client=self.arvrunner.api,
                                                      keep_client=self.arvrunner.keep_client,
                                                      num_retries=self.arvrunner.num_retries)
-                generatemapper = NoFollowPathMapper([self.generatefiles], "", "",
+                generatemapper = NoFollowPathMapper(self.generatefiles["listing"], "", "",
                                                      separateDirs=False)
  
-                sorteditems = sorted(generatemapper.items(), None, key=lambda n: n[1].target)
+                sorteditems = sorted(list(generatemapper.items()), None, key=lambda n: n[1].target)
  
                  logger.debug("generatemapper is %s", sorteditems)
  
@@ -156,8 +173,14 @@ class ArvadosContainer(object):
  
                  keepemptydirs(vwd)
  
-                with Perf(metrics, "generatefiles.save_new %s" % self.name):
-                    vwd.save_new()
+                if not runtimeContext.current_container:
+                    runtimeContext.current_container = arvados_cwl.util.get_current_container(self.arvrunner.api, self.arvrunner.num_retries, logger)
+                info = arvados_cwl.util.get_intermediate_collection_info(self.name, runtimeContext.current_container, runtimeContext.intermediate_output_ttl)
+                vwd.save_new(name=info["name"],
+                             owner_uuid=runtimeContext.project_uuid,
+                             ensure_unique_name=True,
+                             trash_at=info["trash_at"],
+                             properties=info["properties"])
  
                  prev = None
                  for f, p in sorteditems:
@@ -190,23 +213,23 @@ class ArvadosContainer(object):
              mounts["stdout"] = {"kind": "file",
                                  "path": "%s/%s" % (self.outdir, self.stdout)}
  
-        (docker_req, docker_is_req) = get_feature(self, "DockerRequirement")
+        (docker_req, docker_is_req) = self.get_requirement("DockerRequirement")
          if not docker_req:
              docker_req = {"dockerImageId": "arvados/jobs"}
  
          container_request["container_image"] = arv_docker_get_image(self.arvrunner.api,
-                                                                     docker_req,
-                                                                     pull_image,
-                                                                     self.arvrunner.project_uuid)
+                                                                    docker_req,
+                                                                    runtimeContext.pull_image,
+                                                                    runtimeContext.project_uuid)
  
-        api_req, _ = get_feature(self, "http://arvados.org/cwl#APIRequirement")
+        api_req, _ = self.get_requirement("http://arvados.org/cwl#APIRequirement")
          if api_req:
              runtime_constraints["API"] = True
  
-        runtime_req, _ = get_feature(self, "http://arvados.org/cwl#RuntimeConstraints")
+        runtime_req, _ = self.get_requirement("http://arvados.org/cwl#RuntimeConstraints")
          if runtime_req:
              if "keep_cache" in runtime_req:
-                runtime_constraints["keep_cache_ram"] = runtime_req["keep_cache"] * 2**20
+                runtime_constraints["keep_cache_ram"] = math.ceil(runtime_req["keep_cache"] * 2**20)
              if "outputDirType" in runtime_req:
                  if runtime_req["outputDirType"] == "local_output_dir":
                      # Currently the default behavior.
@@ -217,11 +240,11 @@ class ArvadosContainer(object):
                          "writable": True
                      }
  
-        partition_req, _ = get_feature(self, "http://arvados.org/cwl#PartitionRequirement")
+        partition_req, _ = self.get_requirement("http://arvados.org/cwl#PartitionRequirement")
          if partition_req:
              scheduling_parameters["partitions"] = aslist(partition_req["partition"])
  
-        intermediate_output_req, _ = get_feature(self, "http://arvados.org/cwl#IntermediateOutput")
+        intermediate_output_req, _ = self.get_requirement("http://arvados.org/cwl#IntermediateOutput")
          if intermediate_output_req:
              self.output_ttl = intermediate_output_req["outputTTL"]
          else:
@@ -230,21 +253,29 @@ class ArvadosContainer(object):
          if self.output_ttl < 0:
              raise WorkflowException("Invalid value %d for output_ttl, cannot be less than zero" % container_request["output_ttl"])
  
+        if self.timelimit is not None:
+            scheduling_parameters["max_run_time"] = self.timelimit
+
+        extra_submit_params = {}
+        if runtimeContext.submit_runner_cluster:
+            extra_submit_params["cluster_id"] = runtimeContext.submit_runner_cluster
+
+        container_request["output_name"] = "Output for step %s" % (self.name)
          container_request["output_ttl"] = self.output_ttl
          container_request["mounts"] = mounts
          container_request["secret_mounts"] = secret_mounts
          container_request["runtime_constraints"] = runtime_constraints
          container_request["scheduling_parameters"] = scheduling_parameters
  
-        enable_reuse = kwargs.get("enable_reuse", True)
+        enable_reuse = runtimeContext.enable_reuse
          if enable_reuse:
-            reuse_req, _ = get_feature(self, "http://arvados.org/cwl#ReuseRequirement")
+            reuse_req, _ = self.get_requirement("http://arvados.org/cwl#ReuseRequirement")
              if reuse_req:
                  enable_reuse = reuse_req["enableReuse"]
          container_request["use_existing"] = enable_reuse
  
-        if kwargs.get("runnerjob", "").startswith("arvwf:"):
-            wfuuid = kwargs["runnerjob"][6:kwargs["runnerjob"].index("#")]
+        if runtimeContext.runnerjob.startswith("arvwf:"):
+            wfuuid = runtimeContext.runnerjob[6:runtimeContext.runnerjob.index("#")]
              wfrecord = self.arvrunner.api.workflows().get(uuid=wfuuid).execute(num_retries=self.arvrunner.num_retries)
              if container_request["name"] == "main":
                  container_request["name"] = wfrecord["name"]
@@ -253,14 +284,16 @@ class ArvadosContainer(object):
          self.output_callback = self.arvrunner.get_wrapped_callback(self.output_callback)
  
          try:
-            if kwargs.get("submit_request_uuid"):
+            if runtimeContext.submit_request_uuid:
                  response = self.arvrunner.api.container_requests().update(
-                    uuid=kwargs["submit_request_uuid"],
-                    body=container_request
+                    uuid=runtimeContext.submit_request_uuid,
+                    body=container_request,
+                    **extra_submit_params
                  ).execute(num_retries=self.arvrunner.num_retries)
              else:
                  response = self.arvrunner.api.container_requests().create(
-                    body=container_request
+                    body=container_request,
+                    **extra_submit_params
                  ).execute(num_retries=self.arvrunner.num_retries)
  
              self.uuid = response["uuid"]
@@ -300,7 +333,10 @@ class ArvadosContainer(object):
                                                             api_client=self.arvrunner.api,
                                                             keep_client=self.arvrunner.keep_client,
                                                             num_retries=self.arvrunner.num_retries)
-                done.logtail(logc, logger, "%s error log:" % self.arvrunner.label(self))
+                label = self.arvrunner.label(self)
+                done.logtail(
+                    logc, logger.error,
+                    "%s (%s) error log:" % (label, record["uuid"]), maxlen=40)
  
              if record["output_uuid"]:
                  if self.arvrunner.trash_intermediate or self.arvrunner.intermediate_output_ttl:
@@ -329,7 +365,7 @@ class ArvadosContainer(object):
  class RunnerContainer(Runner):
      """Submit and manage a container that runs arvados-cwl-runner."""
  
-    def arvados_job_spec(self, dry_run=False, pull_image=True, **kwargs):
+    def arvados_job_spec(self, runtimeContext):
          """Create an Arvados container request for this workflow.
  
          The returned dict can be used to create a container passed as
@@ -373,16 +409,16 @@ class RunnerContainer(Runner):
              },
              "secret_mounts": secret_mounts,
              "runtime_constraints": {
-                "vcpus": 1,
-                "ram": 1024*1024 * self.submit_runner_ram,
+                "vcpus": math.ceil(self.submit_runner_cores),
+                "ram": 1024*1024 * (math.ceil(self.submit_runner_ram) + math.ceil(self.collection_cache_size)),
                  "API": True
              },
              "use_existing": self.enable_reuse,
              "properties": {}
          }
  
-        if self.tool.tool.get("id", "").startswith("keep:"):
-            sp = self.tool.tool["id"].split('/')
+        if self.embedded_tool.tool.get("id", "").startswith("keep:"):
+            sp = self.embedded_tool.tool["id"].split('/')
              workflowcollection = sp[0][5:]
              workflowname = "/".join(sp[1:])
              workflowpath = "/var/lib/cwl/workflow/%s" % workflowname
@@ -391,14 +427,14 @@ class RunnerContainer(Runner):
                  "portable_data_hash": "%s" % workflowcollection
              }
          else:
-            packed = packed_workflow(self.arvrunner, self.tool, self.merged_map)
+            packed = packed_workflow(self.arvrunner, self.embedded_tool, self.merged_map)
              workflowpath = "/var/lib/cwl/workflow.json#main"
              container_req["mounts"]["/var/lib/cwl/workflow.json"] = {
                  "kind": "json",
                  "content": packed
              }
-            if self.tool.tool.get("id", "").startswith("arvwf:"):
-                container_req["properties"]["template_uuid"] = self.tool.tool["id"][6:33]
+            if self.embedded_tool.tool.get("id", "").startswith("arvwf:"):
+                container_req["properties"]["template_uuid"] = self.embedded_tool.tool["id"][6:33]
  
  
          # --local means execute the workflow instead of submitting a container request
@@ -408,6 +444,7 @@ class RunnerContainer(Runner):
          # --eval-timeout is the timeout for javascript invocation
          # --parallel-task-count is the number of threads to use for job submission
          # --enable/disable-reuse sets desired job reuse
+        # --collection-cache-size sets aside memory to store collections
          command = ["arvados-cwl-runner",
                     "--local",
                     "--api=containers",
@@ -415,7 +452,8 @@ class RunnerContainer(Runner):
                     "--disable-validate",
                     "--eval-timeout=%s" % self.arvrunner.eval_timeout,
                     "--thread-count=%s" % self.arvrunner.thread_count,
-                   "--enable-reuse" if self.enable_reuse else "--disable-reuse"]
+                   "--enable-reuse" if self.enable_reuse else "--disable-reuse",
+                   "--collection-cache-size=%s" % self.collection_cache_size]
  
          if self.output_name:
              command.append("--output-name=" + self.output_name)
@@ -424,9 +462,12 @@ class RunnerContainer(Runner):
          if self.output_tags:
              command.append("--output-tags=" + self.output_tags)
  
-        if kwargs.get("debug"):
+        if runtimeContext.debug:
              command.append("--debug")
  
+        if runtimeContext.storage_classes != "default":
+            command.append("--storage-classes=" + runtimeContext.storage_classes)
+
          if self.on_error:
              command.append("--on-error=" + self.on_error)
  
@@ -446,26 +487,32 @@ class RunnerContainer(Runner):
          return container_req
  
  
-    def run(self, **kwargs):
-        kwargs["keepprefix"] = "keep:"
-        job_spec = self.arvados_job_spec(**kwargs)
+    def run(self, runtimeContext):
+        runtimeContext.keepprefix = "keep:"
+        job_spec = self.arvados_job_spec(runtimeContext)
          if self.arvrunner.project_uuid:
              job_spec["owner_uuid"] = self.arvrunner.project_uuid
  
-        if kwargs.get("submit_request_uuid"):
+        extra_submit_params = {}
+        if runtimeContext.submit_runner_cluster:
+            extra_submit_params["cluster_id"] = runtimeContext.submit_runner_cluster
+
+        if runtimeContext.submit_request_uuid:
              response = self.arvrunner.api.container_requests().update(
-                uuid=kwargs["submit_request_uuid"],
-                body=job_spec
+                uuid=runtimeContext.submit_request_uuid,
+                body=job_spec,
+                **extra_submit_params
              ).execute(num_retries=self.arvrunner.num_retries)
          else:
              response = self.arvrunner.api.container_requests().create(
-                body=job_spec
+                body=job_spec,
+                **extra_submit_params
              ).execute(num_retries=self.arvrunner.num_retries)
  
          self.uuid = response["uuid"]
          self.arvrunner.process_submitted(self)
  
-        logger.info("%s submitted container %s", self.arvrunner.label(self), response["uuid"])
+        logger.info("%s submitted container_request %s", self.arvrunner.label(self), response["uuid"])
  
      def done(self, record):
          try: