9964: A few more special cases of output patterns
[arvados.git] / sdk / cwl / arvados_cwl / arvcontainer.py
1 # Copyright (C) The Arvados Authors. All rights reserved.
2 #
3 # SPDX-License-Identifier: Apache-2.0
4
5 import logging
6 import json
7 import os
8 import urllib.request, urllib.parse, urllib.error
9 import time
10 import datetime
11 import ciso8601
12 import uuid
13 import math
14 import re
15
16 import arvados_cwl.util
17 import ruamel.yaml
18
19 from cwltool.errors import WorkflowException
20 from cwltool.process import UnsupportedRequirement, shortname
21 from cwltool.utils import aslist, adjustFileObjs, adjustDirObjs, visit_class
22 from cwltool.job import JobBase
23
24 import arvados.collection
25
26 import crunchstat_summary.summarizer
27 import crunchstat_summary.reader
28
29 from .arvdocker import arv_docker_get_image
30 from . import done
31 from .runner import Runner, arvados_jobs_image, packed_workflow, trim_anonymous_location, remove_redundant_fields, make_builder
32 from .fsaccess import CollectionFetcher
33 from .pathmapper import NoFollowPathMapper, trim_listing
34 from .perf import Perf
35 from ._version import __version__
36
37 logger = logging.getLogger('arvados.cwl-runner')
38 metrics = logging.getLogger('arvados.cwl-runner.metrics')
39
40 def cleanup_name_for_collection(name):
41     return name.replace("/", " ")
42
43 class ArvadosContainer(JobBase):
44     """Submit and manage a Crunch container request for executing a CWL CommandLineTool."""
45
46     def __init__(self, runner, job_runtime, globpatterns,
47                  builder,   # type: Builder
48                  joborder,  # type: Dict[Text, Union[Dict[Text, Any], List, Text]]
49                  make_path_mapper,  # type: Callable[..., PathMapper]
50                  requirements,      # type: List[Dict[Text, Text]]
51                  hints,     # type: List[Dict[Text, Text]]
52                  name       # type: Text
53     ):
54         super(ArvadosContainer, self).__init__(builder, joborder, make_path_mapper, requirements, hints, name)
55         self.arvrunner = runner
56         self.job_runtime = job_runtime
57         self.running = False
58         self.uuid = None
59         self.attempt_count = 0
60         self.globpatterns = globpatterns
61
62     def update_pipeline_component(self, r):
63         pass
64
65     def _required_env(self):
66         env = {}
67         env["HOME"] = self.outdir
68         env["TMPDIR"] = self.tmpdir
69         return env
70
71     def run(self, toplevelRuntimeContext):
72         # ArvadosCommandTool subclasses from cwltool.CommandLineTool,
73         # which calls makeJobRunner() to get a new ArvadosContainer
74         # object.  The fields that define execution such as
75         # command_line, environment, etc are set on the
76         # ArvadosContainer object by CommandLineTool.job() before
77         # run() is called.
78
79         runtimeContext = self.job_runtime
80
81         if runtimeContext.submit_request_uuid:
82             container_request = self.arvrunner.api.container_requests().get(
83                 uuid=runtimeContext.submit_request_uuid
84             ).execute(num_retries=self.arvrunner.num_retries)
85         else:
86             container_request = {}
87
88         container_request["command"] = self.command_line
89         container_request["name"] = self.name
90         container_request["output_path"] = self.outdir
91         container_request["cwd"] = self.outdir
92         container_request["priority"] = runtimeContext.priority
93         container_request["state"] = "Uncommitted"
94         container_request.setdefault("properties", {})
95
96         container_request["properties"]["cwl_input"] = self.joborder
97
98         runtime_constraints = {}
99
100         if runtimeContext.project_uuid:
101             container_request["owner_uuid"] = runtimeContext.project_uuid
102
103         if self.arvrunner.secret_store.has_secret(self.command_line):
104             raise WorkflowException("Secret material leaked on command line, only file literals may contain secrets")
105
106         if self.arvrunner.secret_store.has_secret(self.environment):
107             raise WorkflowException("Secret material leaked in environment, only file literals may contain secrets")
108
109         resources = self.builder.resources
110         if resources is not None:
111             runtime_constraints["vcpus"] = math.ceil(resources.get("cores", 1))
112             runtime_constraints["ram"] = math.ceil(resources.get("ram") * 2**20)
113
114         mounts = {
115             self.outdir: {
116                 "kind": "tmp",
117                 "capacity": math.ceil(resources.get("outdirSize", 0) * 2**20)
118             },
119             self.tmpdir: {
120                 "kind": "tmp",
121                 "capacity": math.ceil(resources.get("tmpdirSize", 0) * 2**20)
122             }
123         }
124         secret_mounts = {}
125         scheduling_parameters = {}
126
127         rf = [self.pathmapper.mapper(f) for f in self.pathmapper.referenced_files]
128         rf.sort(key=lambda k: k.resolved)
129         prevdir = None
130         for resolved, target, tp, stg in rf:
131             if not stg:
132                 continue
133             if prevdir and target.startswith(prevdir):
134                 continue
135             if tp == "Directory":
136                 targetdir = target
137             else:
138                 targetdir = os.path.dirname(target)
139             sp = resolved.split("/", 1)
140             pdh = sp[0][5:]   # remove "keep:"
141             mounts[targetdir] = {
142                 "kind": "collection",
143                 "portable_data_hash": pdh
144             }
145             if pdh in self.pathmapper.pdh_to_uuid:
146                 mounts[targetdir]["uuid"] = self.pathmapper.pdh_to_uuid[pdh]
147             if len(sp) == 2:
148                 if tp == "Directory":
149                     path = sp[1]
150                 else:
151                     path = os.path.dirname(sp[1])
152                 if path and path != "/":
153                     mounts[targetdir]["path"] = path
154             prevdir = targetdir + "/"
155
156         intermediate_collection_info = arvados_cwl.util.get_intermediate_collection_info(self.name, runtimeContext.current_container, runtimeContext.intermediate_output_ttl)
157
158         with Perf(metrics, "generatefiles %s" % self.name):
159             if self.generatefiles["listing"]:
160                 vwd = arvados.collection.Collection(api_client=self.arvrunner.api,
161                                                     keep_client=self.arvrunner.keep_client,
162                                                     num_retries=self.arvrunner.num_retries)
163                 generatemapper = NoFollowPathMapper(self.generatefiles["listing"], "", "",
164                                                     separateDirs=False)
165
166                 sorteditems = sorted(generatemapper.items(), key=lambda n: n[1].target)
167
168                 logger.debug("generatemapper is %s", sorteditems)
169
170                 with Perf(metrics, "createfiles %s" % self.name):
171                     for f, p in sorteditems:
172                         if not p.target:
173                             continue
174
175                         if p.target.startswith("/"):
176                             dst = p.target[len(self.outdir)+1:] if p.target.startswith(self.outdir+"/") else p.target[1:]
177                         else:
178                             dst = p.target
179
180                         if p.type in ("File", "Directory", "WritableFile", "WritableDirectory"):
181                             if p.resolved.startswith("_:"):
182                                 vwd.mkdirs(dst)
183                             else:
184                                 source, path = self.arvrunner.fs_access.get_collection(p.resolved)
185                                 vwd.copy(path or ".", dst, source_collection=source)
186                         elif p.type == "CreateFile":
187                             if self.arvrunner.secret_store.has_secret(p.resolved):
188                                 mountpoint = p.target if p.target.startswith("/") else os.path.join(self.outdir, p.target)
189                                 secret_mounts[mountpoint] = {
190                                     "kind": "text",
191                                     "content": self.arvrunner.secret_store.retrieve(p.resolved)
192                                 }
193                             else:
194                                 with vwd.open(dst, "w") as n:
195                                     n.write(p.resolved)
196
197                 def keepemptydirs(p):
198                     if isinstance(p, arvados.collection.RichCollectionBase):
199                         if len(p) == 0:
200                             p.open(".keep", "w").close()
201                         else:
202                             for c in p:
203                                 keepemptydirs(p[c])
204
205                 keepemptydirs(vwd)
206
207                 if not runtimeContext.current_container:
208                     runtimeContext.current_container = arvados_cwl.util.get_current_container(self.arvrunner.api, self.arvrunner.num_retries, logger)
209                 vwd.save_new(name=intermediate_collection_info["name"],
210                              owner_uuid=runtimeContext.project_uuid,
211                              ensure_unique_name=True,
212                              trash_at=intermediate_collection_info["trash_at"],
213                              properties=intermediate_collection_info["properties"])
214
215                 prev = None
216                 for f, p in sorteditems:
217                     if (not p.target or self.arvrunner.secret_store.has_secret(p.resolved) or
218                         (prev is not None and p.target.startswith(prev))):
219                         continue
220                     if p.target.startswith("/"):
221                         dst = p.target[len(self.outdir)+1:] if p.target.startswith(self.outdir+"/") else p.target[1:]
222                     else:
223                         dst = p.target
224                     mountpoint = p.target if p.target.startswith("/") else os.path.join(self.outdir, p.target)
225                     mounts[mountpoint] = {"kind": "collection",
226                                           "portable_data_hash": vwd.portable_data_hash(),
227                                           "path": dst}
228                     if p.type.startswith("Writable"):
229                         mounts[mountpoint]["writable"] = True
230                     prev = p.target + "/"
231
232         container_request["environment"] = {"TMPDIR": self.tmpdir, "HOME": self.outdir}
233         if self.environment:
234             container_request["environment"].update(self.environment)
235
236         if self.stdin:
237             sp = self.stdin[6:].split("/", 1)
238             mounts["stdin"] = {"kind": "collection",
239                                 "portable_data_hash": sp[0],
240                                 "path": sp[1]}
241
242         if self.stderr:
243             mounts["stderr"] = {"kind": "file",
244                                 "path": "%s/%s" % (self.outdir, self.stderr)}
245
246         if self.stdout:
247             mounts["stdout"] = {"kind": "file",
248                                 "path": "%s/%s" % (self.outdir, self.stdout)}
249
250         (docker_req, docker_is_req) = self.get_requirement("DockerRequirement")
251
252         container_request["container_image"] = arv_docker_get_image(self.arvrunner.api,
253                                                                     docker_req,
254                                                                     runtimeContext.pull_image,
255                                                                     runtimeContext)
256
257         network_req, _ = self.get_requirement("NetworkAccess")
258         if network_req:
259             runtime_constraints["API"] = network_req["networkAccess"]
260
261         api_req, _ = self.get_requirement("http://arvados.org/cwl#APIRequirement")
262         if api_req:
263             runtime_constraints["API"] = True
264
265         use_disk_cache = (self.arvrunner.api.config()["Containers"].get("DefaultKeepCacheRAM", 0) == 0)
266
267         keep_cache_type_req, _ = self.get_requirement("http://arvados.org/cwl#KeepCacheTypeRequirement")
268         if keep_cache_type_req:
269             if "keepCacheType" in keep_cache_type_req:
270                 if keep_cache_type_req["keepCacheType"] == "ram_cache":
271                     use_disk_cache = False
272
273         runtime_req, _ = self.get_requirement("http://arvados.org/cwl#RuntimeConstraints")
274         if runtime_req:
275             if "keep_cache" in runtime_req:
276                 if use_disk_cache:
277                     # If DefaultKeepCacheRAM is zero it means we should use disk cache.
278                     runtime_constraints["keep_cache_disk"] = math.ceil(runtime_req["keep_cache"] * 2**20)
279                 else:
280                     runtime_constraints["keep_cache_ram"] = math.ceil(runtime_req["keep_cache"] * 2**20)
281             if "outputDirType" in runtime_req:
282                 if runtime_req["outputDirType"] == "local_output_dir":
283                     # Currently the default behavior.
284                     pass
285                 elif runtime_req["outputDirType"] == "keep_output_dir":
286                     mounts[self.outdir]= {
287                         "kind": "collection",
288                         "writable": True
289                     }
290
291         partition_req, _ = self.get_requirement("http://arvados.org/cwl#PartitionRequirement")
292         if partition_req:
293             scheduling_parameters["partitions"] = aslist(partition_req["partition"])
294
295         intermediate_output_req, _ = self.get_requirement("http://arvados.org/cwl#IntermediateOutput")
296         if intermediate_output_req:
297             self.output_ttl = intermediate_output_req["outputTTL"]
298         else:
299             self.output_ttl = self.arvrunner.intermediate_output_ttl
300
301         if self.output_ttl < 0:
302             raise WorkflowException("Invalid value %d for output_ttl, cannot be less than zero" % container_request["output_ttl"])
303
304
305         if self.arvrunner.api._rootDesc["revision"] >= "20210628":
306             storage_class_req, _ = self.get_requirement("http://arvados.org/cwl#OutputStorageClass")
307             if storage_class_req and storage_class_req.get("intermediateStorageClass"):
308                 container_request["output_storage_classes"] = aslist(storage_class_req["intermediateStorageClass"])
309             else:
310                 container_request["output_storage_classes"] = runtimeContext.intermediate_storage_classes.strip().split(",")
311
312         cuda_req, _ = self.get_requirement("http://commonwl.org/cwltool#CUDARequirement")
313         if cuda_req:
314             runtime_constraints["cuda"] = {
315                 "device_count": resources.get("cudaDeviceCount", 1),
316                 "driver_version": cuda_req["cudaVersionMin"],
317                 "hardware_capability": aslist(cuda_req["cudaComputeCapability"])[0]
318             }
319
320         if runtimeContext.enable_preemptible is False:
321             scheduling_parameters["preemptible"] = False
322         else:
323             preemptible_req, _ = self.get_requirement("http://arvados.org/cwl#UsePreemptible")
324             if preemptible_req:
325                 scheduling_parameters["preemptible"] = preemptible_req["usePreemptible"]
326             elif runtimeContext.enable_preemptible is True:
327                 scheduling_parameters["preemptible"] = True
328             elif runtimeContext.enable_preemptible is None:
329                 pass
330
331         if self.timelimit is not None and self.timelimit > 0:
332             scheduling_parameters["max_run_time"] = self.timelimit
333
334         extra_submit_params = {}
335         if runtimeContext.submit_runner_cluster:
336             extra_submit_params["cluster_id"] = runtimeContext.submit_runner_cluster
337
338         container_request["output_name"] = cleanup_name_for_collection("Output from step %s" % (self.name))
339         container_request["output_ttl"] = self.output_ttl
340         container_request["mounts"] = mounts
341         container_request["secret_mounts"] = secret_mounts
342         container_request["runtime_constraints"] = runtime_constraints
343         container_request["scheduling_parameters"] = scheduling_parameters
344
345         enable_reuse = runtimeContext.enable_reuse
346         if enable_reuse:
347             reuse_req, _ = self.get_requirement("WorkReuse")
348             if reuse_req:
349                 enable_reuse = reuse_req["enableReuse"]
350             reuse_req, _ = self.get_requirement("http://arvados.org/cwl#ReuseRequirement")
351             if reuse_req:
352                 enable_reuse = reuse_req["enableReuse"]
353         container_request["use_existing"] = enable_reuse
354
355         properties_req, _ = self.get_requirement("http://arvados.org/cwl#ProcessProperties")
356         if properties_req:
357             for pr in properties_req["processProperties"]:
358                 container_request["properties"][pr["propertyName"]] = self.builder.do_eval(pr["propertyValue"])
359
360         output_properties_req, _ = self.get_requirement("http://arvados.org/cwl#OutputCollectionProperties")
361         if output_properties_req:
362             if self.arvrunner.api._rootDesc["revision"] >= "20220510":
363                 container_request["output_properties"] = {}
364                 for pr in output_properties_req["outputProperties"]:
365                     container_request["output_properties"][pr["propertyName"]] = self.builder.do_eval(pr["propertyValue"])
366             else:
367                 logger.warning("%s API revision is %s, revision %s is required to support setting properties on output collections.",
368                                self.arvrunner.label(self), self.arvrunner.api._rootDesc["revision"], "20220510")
369
370         if self.arvrunner.api._rootDesc["revision"] >= "20240502" and self.globpatterns:
371             output_glob = []
372             for gb in self.globpatterns:
373                 gb = self.builder.do_eval(gb)
374                 if not gb:
375                     continue
376                 for gbeval in aslist(gb):
377                     if gbeval.startswith(self.outdir+"/"):
378                         gbeval = gbeval[len(self.outdir)+1:]
379                     if gbeval in (self.outdir, "", "."):
380                         output_glob.append("**")
381                     else:
382                         output_glob.append(gbeval)
383                         output_glob.append(gbeval + "/**")
384
385             if "**" in output_glob:
386                 # if it's going to match all, prefer not to provide it
387                 # at all.
388                 output_glob.clear()
389
390             if output_glob:
391                 # Tools should either use cwl.output.json or output
392                 # binding glob. Unfortunately one CWL conformance test
393                 # has both, but that test uses cwl.output.json return
394                 # a string, not a File.
395                 #
396                 # It could happen that a tool uses cwl.output.json,
397                 # references a file, but also uses a glob, and that
398                 # would break, because we don't know about it ahead of
399                 # time.  However I don't think any conformance tests
400                 # do this and I'd even be inclined to rule that out of
401                 # order in the CWL spec if it came up.
402                 output_glob.append("cwl.output.json")
403                 container_request["output_glob"] = output_glob
404
405         ram_multiplier = [1]
406
407         oom_retry_req, _ = self.get_requirement("http://arvados.org/cwl#OutOfMemoryRetry")
408         if oom_retry_req:
409             if oom_retry_req.get('memoryRetryMultiplier'):
410                 ram_multiplier.append(oom_retry_req.get('memoryRetryMultiplier'))
411             elif oom_retry_req.get('memoryRetryMultipler'):
412                 ram_multiplier.append(oom_retry_req.get('memoryRetryMultipler'))
413             else:
414                 ram_multiplier.append(2)
415
416         if runtimeContext.runnerjob.startswith("arvwf:"):
417             wfuuid = runtimeContext.runnerjob[6:runtimeContext.runnerjob.index("#")]
418             wfrecord = self.arvrunner.api.workflows().get(uuid=wfuuid).execute(num_retries=self.arvrunner.num_retries)
419             if container_request["name"] == "main":
420                 container_request["name"] = wfrecord["name"]
421             container_request["properties"]["template_uuid"] = wfuuid
422
423         if self.attempt_count == 0:
424             self.output_callback = self.arvrunner.get_wrapped_callback(self.output_callback)
425
426         try:
427             ram = runtime_constraints["ram"]
428
429             self.uuid = runtimeContext.submit_request_uuid
430
431             for i in ram_multiplier:
432                 runtime_constraints["ram"] = ram * i
433
434                 if self.uuid:
435                     response = self.arvrunner.api.container_requests().update(
436                         uuid=self.uuid,
437                         body=container_request,
438                         **extra_submit_params
439                     ).execute(num_retries=self.arvrunner.num_retries)
440                 else:
441                     response = self.arvrunner.api.container_requests().create(
442                         body=container_request,
443                         **extra_submit_params
444                     ).execute(num_retries=self.arvrunner.num_retries)
445                     self.uuid = response["uuid"]
446
447                 if response["container_uuid"] is not None:
448                     break
449
450             if response["container_uuid"] is None:
451                 runtime_constraints["ram"] = ram * ram_multiplier[self.attempt_count]
452
453             container_request["state"] = "Committed"
454             response = self.arvrunner.api.container_requests().update(
455                 uuid=self.uuid,
456                 body=container_request,
457                 **extra_submit_params
458             ).execute(num_retries=self.arvrunner.num_retries)
459
460             self.arvrunner.process_submitted(self)
461             self.attempt_count += 1
462
463             if response["state"] == "Final":
464                 logger.info("%s reused container %s", self.arvrunner.label(self), response["container_uuid"])
465             else:
466                 logger.info("%s %s state is %s", self.arvrunner.label(self), response["uuid"], response["state"])
467         except Exception as e:
468             logger.exception("%s error submitting container\n%s", self.arvrunner.label(self), e)
469             logger.debug("Container request was %s", container_request)
470             self.output_callback({}, "permanentFail")
471
472     def out_of_memory_retry(self, record, container):
473         oom_retry_req, _ = self.get_requirement("http://arvados.org/cwl#OutOfMemoryRetry")
474         if oom_retry_req is None:
475             return False
476
477         # Sometimes it gets killed with no warning
478         if container["exit_code"] == 137:
479             return True
480
481         logc = arvados.collection.CollectionReader(record["log_uuid"],
482                                                    api_client=self.arvrunner.api,
483                                                    keep_client=self.arvrunner.keep_client,
484                                                    num_retries=self.arvrunner.num_retries)
485
486         loglines = [""]
487         def callback(v1, v2, v3):
488             loglines[0] = v3
489
490         done.logtail(logc, callback, "", maxlen=1000)
491
492         # Check allocation failure
493         oom_matches = oom_retry_req.get('memoryErrorRegex') or r'(bad_alloc|out ?of ?memory|memory ?error|container using over 9.% of memory)'
494         if re.search(oom_matches, loglines[0], re.IGNORECASE | re.MULTILINE):
495             return True
496
497         return False
498
499     def done(self, record):
500         outputs = {}
501         retried = False
502         rcode = None
503         try:
504             container = self.arvrunner.api.containers().get(
505                 uuid=record["container_uuid"]
506             ).execute(num_retries=self.arvrunner.num_retries)
507             if container["state"] == "Complete":
508                 rcode = container["exit_code"]
509                 if self.successCodes and rcode in self.successCodes:
510                     processStatus = "success"
511                 elif self.temporaryFailCodes and rcode in self.temporaryFailCodes:
512                     processStatus = "temporaryFail"
513                 elif self.permanentFailCodes and rcode in self.permanentFailCodes:
514                     processStatus = "permanentFail"
515                 elif rcode == 0:
516                     processStatus = "success"
517                 else:
518                     processStatus = "permanentFail"
519
520                 if processStatus == "permanentFail" and self.attempt_count == 1 and self.out_of_memory_retry(record, container):
521                     logger.warning("%s Container failed with out of memory error, retrying with more RAM.",
522                                  self.arvrunner.label(self))
523                     self.job_runtime.submit_request_uuid = None
524                     self.uuid = None
525                     self.run(None)
526                     retried = True
527                     return
528
529                 if rcode == 137:
530                     logger.warning("%s Container may have been killed for using too much RAM.  Try resubmitting with a higher 'ramMin' or use the arv:OutOfMemoryRetry feature.",
531                                  self.arvrunner.label(self))
532             else:
533                 processStatus = "permanentFail"
534
535             logc = None
536             if record["log_uuid"]:
537                 logc = arvados.collection.Collection(record["log_uuid"],
538                                                      api_client=self.arvrunner.api,
539                                                      keep_client=self.arvrunner.keep_client,
540                                                      num_retries=self.arvrunner.num_retries)
541
542             if processStatus == "permanentFail" and logc is not None:
543                 label = self.arvrunner.label(self)
544                 done.logtail(
545                     logc, logger.error,
546                     "%s (%s) error log:" % (label, record["uuid"]), maxlen=40, include_crunchrun=(rcode is None or rcode > 127))
547
548             if record["output_uuid"]:
549                 if self.arvrunner.trash_intermediate or self.arvrunner.intermediate_output_ttl:
550                     # Compute the trash time to avoid requesting the collection record.
551                     trash_at = ciso8601.parse_datetime_as_naive(record["modified_at"]) + datetime.timedelta(0, self.arvrunner.intermediate_output_ttl)
552                     aftertime = " at %s" % trash_at.strftime("%Y-%m-%d %H:%M:%S UTC") if self.arvrunner.intermediate_output_ttl else ""
553                     orpart = ", or" if self.arvrunner.trash_intermediate and self.arvrunner.intermediate_output_ttl else ""
554                     oncomplete = " upon successful completion of the workflow" if self.arvrunner.trash_intermediate else ""
555                     logger.info("%s Intermediate output %s (%s) will be trashed%s%s%s." % (
556                         self.arvrunner.label(self), record["output_uuid"], container["output"], aftertime, orpart, oncomplete))
557                 self.arvrunner.add_intermediate_output(record["output_uuid"])
558
559             if container["output"]:
560                 outputs = done.done_outputs(self, container, "/tmp", self.outdir, "/keep")
561
562             properties = record["properties"].copy()
563             properties["cwl_output"] = outputs
564             self.arvrunner.api.container_requests().update(
565                 uuid=self.uuid,
566                 body={"container_request": {"properties": properties}}
567             ).execute(num_retries=self.arvrunner.num_retries)
568
569             if logc is not None and self.job_runtime.enable_usage_report is not False:
570                 try:
571                     summarizer = crunchstat_summary.summarizer.ContainerRequestSummarizer(
572                         record,
573                         collection_object=logc,
574                         label=self.name,
575                         arv=self.arvrunner.api)
576                     summarizer.run()
577                     with logc.open("usage_report.html", "wt") as mr:
578                         mr.write(summarizer.html_report())
579                     logc.save()
580
581                     # Post warnings about nodes that are under-utilized.
582                     for rc in summarizer._recommend_gen(lambda x: x):
583                         self.job_runtime.usage_report_notes.append(rc)
584
585                 except Exception as e:
586                     logger.warning("%s unable to generate resource usage report",
587                                  self.arvrunner.label(self),
588                                  exc_info=(e if self.arvrunner.debug else False))
589
590         except WorkflowException as e:
591             # Only include a stack trace if in debug mode.
592             # A stack trace may obfuscate more useful output about the workflow.
593             logger.error("%s unable to collect output from %s:\n%s",
594                          self.arvrunner.label(self), container["output"], e, exc_info=(e if self.arvrunner.debug else False))
595             processStatus = "permanentFail"
596         except Exception:
597             logger.exception("%s while getting output object:", self.arvrunner.label(self))
598             processStatus = "permanentFail"
599         finally:
600             if not retried:
601                 self.output_callback(outputs, processStatus)
602
603
604 class RunnerContainer(Runner):
605     """Submit and manage a container that runs arvados-cwl-runner."""
606
607     def arvados_job_spec(self, runtimeContext, git_info):
608         """Create an Arvados container request for this workflow.
609
610         The returned dict can be used to create a container passed as
611         the +body+ argument to container_requests().create().
612         """
613
614         adjustDirObjs(self.job_order, trim_listing)
615         visit_class(self.job_order, ("File", "Directory"), trim_anonymous_location)
616         visit_class(self.job_order, ("File", "Directory"), remove_redundant_fields)
617
618         secret_mounts = {}
619         for param in sorted(self.job_order.keys()):
620             if self.secret_store.has_secret(self.job_order[param]):
621                 mnt = "/secrets/s%d" % len(secret_mounts)
622                 secret_mounts[mnt] = {
623                     "kind": "text",
624                     "content": self.secret_store.retrieve(self.job_order[param])
625                 }
626                 self.job_order[param] = {"$include": mnt}
627
628         container_image = arvados_jobs_image(self.arvrunner, self.jobs_image, runtimeContext)
629
630         workflow_runner_req, _ = self.embedded_tool.get_requirement("http://arvados.org/cwl#WorkflowRunnerResources")
631         if workflow_runner_req and workflow_runner_req.get("acrContainerImage"):
632             container_image = workflow_runner_req.get("acrContainerImage")
633
634         container_req = {
635             "name": self.name,
636             "output_path": "/var/spool/cwl",
637             "cwd": "/var/spool/cwl",
638             "priority": self.priority,
639             "state": "Committed",
640             "container_image": container_image,
641             "mounts": {
642                 "/var/lib/cwl/cwl.input.json": {
643                     "kind": "json",
644                     "content": self.job_order
645                 },
646                 "stdout": {
647                     "kind": "file",
648                     "path": "/var/spool/cwl/cwl.output.json"
649                 },
650                 "/var/spool/cwl": {
651                     "kind": "collection",
652                     "writable": True
653                 }
654             },
655             "secret_mounts": secret_mounts,
656             "runtime_constraints": {
657                 "vcpus": math.ceil(self.submit_runner_cores),
658                 "ram": 1024*1024 * (math.ceil(self.submit_runner_ram) + math.ceil(self.collection_cache_size)),
659                 "API": True
660             },
661             "use_existing": self.reuse_runner,
662             "properties": {}
663         }
664
665         if self.embedded_tool.tool.get("id", "").startswith("keep:"):
666             sp = self.embedded_tool.tool["id"].split('/')
667             workflowcollection = sp[0][5:]
668             workflowname = "/".join(sp[1:])
669             workflowpath = "/var/lib/cwl/workflow/%s" % workflowname
670             container_req["mounts"]["/var/lib/cwl/workflow"] = {
671                 "kind": "collection",
672                 "portable_data_hash": "%s" % workflowcollection
673             }
674         elif self.embedded_tool.tool.get("id", "").startswith("arvwf:"):
675             uuid, frg = urllib.parse.urldefrag(self.embedded_tool.tool["id"])
676             workflowpath = "/var/lib/cwl/workflow.json#" + frg
677             packedtxt = self.loadingContext.loader.fetch_text(uuid)
678             yaml = ruamel.yaml.YAML(typ='safe', pure=True)
679             packed = yaml.load(packedtxt)
680             container_req["mounts"]["/var/lib/cwl/workflow.json"] = {
681                 "kind": "json",
682                 "content": packed
683             }
684             container_req["properties"]["template_uuid"] = self.embedded_tool.tool["id"][6:33]
685         elif self.embedded_tool.tool.get("id", "").startswith("file:"):
686             raise WorkflowException("Tool id '%s' is a local file but expected keep: or arvwf:" % self.embedded_tool.tool.get("id"))
687         else:
688             main = self.loadingContext.loader.idx["_:main"]
689             if main.get("id") == "_:main":
690                 del main["id"]
691             workflowpath = "/var/lib/cwl/workflow.json#main"
692             container_req["mounts"]["/var/lib/cwl/workflow.json"] = {
693                 "kind": "json",
694                 "content": main
695             }
696
697         container_req["properties"].update({k.replace("http://arvados.org/cwl#", "arv:"): v for k, v in git_info.items()})
698
699         properties_req, _ = self.embedded_tool.get_requirement("http://arvados.org/cwl#ProcessProperties")
700         if properties_req:
701             builder = make_builder(self.job_order, self.embedded_tool.hints, self.embedded_tool.requirements, runtimeContext, self.embedded_tool.metadata)
702             for pr in properties_req["processProperties"]:
703                 container_req["properties"][pr["propertyName"]] = builder.do_eval(pr["propertyValue"])
704
705         # --local means execute the workflow instead of submitting a container request
706         # --api=containers means use the containers API
707         # --no-log-timestamps means don't add timestamps (the logging infrastructure does this)
708         # --disable-validate because we already validated so don't need to do it again
709         # --eval-timeout is the timeout for javascript invocation
710         # --parallel-task-count is the number of threads to use for job submission
711         # --enable/disable-reuse sets desired job reuse
712         # --collection-cache-size sets aside memory to store collections
713         command = ["arvados-cwl-runner",
714                    "--local",
715                    "--api=containers",
716                    "--no-log-timestamps",
717                    "--disable-validate",
718                    "--disable-color",
719                    "--eval-timeout=%s" % self.arvrunner.eval_timeout,
720                    "--thread-count=%s" % self.arvrunner.thread_count,
721                    "--enable-reuse" if self.enable_reuse else "--disable-reuse",
722                    "--collection-cache-size=%s" % self.collection_cache_size]
723
724         if self.output_name:
725             command.append("--output-name=" + self.output_name)
726             container_req["output_name"] = self.output_name
727
728         if self.output_tags:
729             command.append("--output-tags=" + self.output_tags)
730
731         if runtimeContext.debug:
732             command.append("--debug")
733
734         if runtimeContext.storage_classes != "default" and runtimeContext.storage_classes:
735             command.append("--storage-classes=" + runtimeContext.storage_classes)
736
737         if runtimeContext.intermediate_storage_classes != "default" and runtimeContext.intermediate_storage_classes:
738             command.append("--intermediate-storage-classes=" + runtimeContext.intermediate_storage_classes)
739
740         if runtimeContext.on_error:
741             command.append("--on-error=" + self.on_error)
742
743         if runtimeContext.intermediate_output_ttl:
744             command.append("--intermediate-output-ttl=%d" % runtimeContext.intermediate_output_ttl)
745
746         if runtimeContext.trash_intermediate:
747             command.append("--trash-intermediate")
748
749         if runtimeContext.project_uuid:
750             command.append("--project-uuid="+runtimeContext.project_uuid)
751
752         if self.enable_dev:
753             command.append("--enable-dev")
754
755         if runtimeContext.enable_preemptible is True:
756             command.append("--enable-preemptible")
757
758         if runtimeContext.enable_preemptible is False:
759             command.append("--disable-preemptible")
760
761         if runtimeContext.varying_url_params:
762             command.append("--varying-url-params="+runtimeContext.varying_url_params)
763
764         if runtimeContext.prefer_cached_downloads:
765             command.append("--prefer-cached-downloads")
766
767         if runtimeContext.enable_usage_report is True:
768             command.append("--enable-usage-report")
769
770         if runtimeContext.enable_usage_report is False:
771             command.append("--disable-usage-report")
772
773         if self.fast_parser:
774             command.append("--fast-parser")
775
776         command.extend([workflowpath, "/var/lib/cwl/cwl.input.json"])
777
778         container_req["command"] = command
779
780         return container_req
781
782
783     def run(self, runtimeContext):
784         runtimeContext.keepprefix = "keep:"
785         job_spec = self.arvados_job_spec(runtimeContext, self.git_info)
786         if runtimeContext.project_uuid:
787             job_spec["owner_uuid"] = runtimeContext.project_uuid
788
789         extra_submit_params = {}
790         if runtimeContext.submit_runner_cluster:
791             extra_submit_params["cluster_id"] = runtimeContext.submit_runner_cluster
792
793         if runtimeContext.submit_request_uuid:
794             if "cluster_id" in extra_submit_params:
795                 # Doesn't make sense for "update" and actually fails
796                 del extra_submit_params["cluster_id"]
797             response = self.arvrunner.api.container_requests().update(
798                 uuid=runtimeContext.submit_request_uuid,
799                 body=job_spec,
800                 **extra_submit_params
801             ).execute(num_retries=self.arvrunner.num_retries)
802         else:
803             response = self.arvrunner.api.container_requests().create(
804                 body=job_spec,
805                 **extra_submit_params
806             ).execute(num_retries=self.arvrunner.num_retries)
807
808         self.uuid = response["uuid"]
809         self.arvrunner.process_submitted(self)
810
811         logger.info("%s submitted container_request %s", self.arvrunner.label(self), response["uuid"])
812
813         workbench2 = self.arvrunner.api.config()["Services"]["Workbench2"]["ExternalURL"]
814         if workbench2:
815             url = "{}processes/{}".format(workbench2, response["uuid"])
816             logger.info("Monitor workflow progress at %s", url)
817
818
819     def done(self, record):
820         try:
821             container = self.arvrunner.api.containers().get(
822                 uuid=record["container_uuid"]
823             ).execute(num_retries=self.arvrunner.num_retries)
824             container["log"] = record["log_uuid"]
825         except Exception:
826             logger.exception("%s while getting runner container", self.arvrunner.label(self))
827             self.arvrunner.output_callback({}, "permanentFail")
828         else:
829             super(RunnerContainer, self).done(container)