11684: Reverted easy fix to expose the bug: when there's a delay writing a block...

[arvados.git] / sdk / python / arvados / collection.py
diff --git a/sdk/python/arvados/collection.py b/sdk/python/arvados/collection.py

index 1a427814cf4d5bc13ffbeca75f7c22c87134962c..77312e4d4917a276f00b46e90e13ab13ba0d5ac4 100644 (file)
--- a/sdk/python/arvados/collection.py
+++ b/sdk/python/arvados/collection.py
@@ -1,4 +1,8 @@
  from __future__ import absolute_import
+from future.utils import listitems, listvalues, viewkeys
+from builtins import str
+from past.builtins import basestring
+from builtins import object
  import functools
  import logging
  import os
@@ -217,7 +221,11 @@ class CollectionWriter(CollectionBase):
          self.do_queued_work()
  
      def write(self, newdata):
-        if hasattr(newdata, '__iter__'):
+        if isinstance(newdata, bytes):
+            pass
+        elif isinstance(newdata, str):
+            newdata = newdata.encode()
+        elif hasattr(newdata, '__iter__'):
              for s in newdata:
                  self.write(s)
              return
@@ -257,7 +265,7 @@ class CollectionWriter(CollectionBase):
          return self._last_open
  
      def flush_data(self):
-        data_buffer = ''.join(self._data_buffer)
+        data_buffer = b''.join(self._data_buffer)
          if data_buffer:
              self._current_stream_locators.append(
                  self._my_keep().put(
@@ -347,11 +355,12 @@ class CollectionWriter(CollectionBase):
          sending manifest_text() to the API server's "create
          collection" endpoint.
          """
-        return self._my_keep().put(self.manifest_text(), copies=self.replication)
+        return self._my_keep().put(self.manifest_text().encode(),
+                                   copies=self.replication)
  
      def portable_data_hash(self):
-        stripped = self.stripped_manifest()
-        return hashlib.md5(stripped).hexdigest() + '+' + str(len(stripped))
+        stripped = self.stripped_manifest().encode()
+        return '{}+{}'.format(hashlib.md5(stripped).hexdigest(), len(stripped))
  
      def manifest_text(self):
          self.finish_current_stream()
@@ -419,7 +428,7 @@ class ResumableCollectionWriter(CollectionWriter):
          return writer
  
      def check_dependencies(self):
-        for path, orig_stat in self._dependencies.items():
+        for path, orig_stat in listitems(self._dependencies):
              if not S_ISREG(orig_stat[ST_MODE]):
                  raise errors.StaleWriterStateError("{} not file".format(path))
              try:
@@ -613,7 +622,12 @@ class RichCollectionBase(CollectionBase):
          :path:
            path to a file in the collection
          :mode:
-          one of "r", "r+", "w", "w+", "a", "a+"
+          a string consisting of "r", "w", or "a", optionally followed
+          by "b" or "t", optionally followed by "+".
+          :"b":
+            binary mode: write() accepts bytes, read() returns bytes.
+          :"t":
+            text mode (default): write() accepts strings, read() returns strings.
            :"r":
              opens for reading
            :"r+":
@@ -625,33 +639,28 @@ class RichCollectionBase(CollectionBase):
              the end of the file.  Writing does not affect the file pointer for
              reading.
          """
-        mode = mode.replace("b", "")
-        if len(mode) == 0 or mode[0] not in ("r", "w", "a"):
-            raise errors.ArgumentError("Bad mode '%s'" % mode)
-        create = (mode != "r")
  
-        if create and not self.writable():
-            raise IOError(errno.EROFS, "Collection is read only")
+        if not re.search(r'^[rwa][bt]?\+?$', mode):
+            raise errors.ArgumentError("Invalid mode {!r}".format(mode))
  
-        if create:
-            arvfile = self.find_or_create(path, FILE)
-        else:
+        if mode[0] == 'r' and '+' not in mode:
+            fclass = ArvadosFileReader
              arvfile = self.find(path)
+        elif not self.writable():
+            raise IOError(errno.EROFS, "Collection is read only")
+        else:
+            fclass = ArvadosFileWriter
+            arvfile = self.find_or_create(path, FILE)
  
          if arvfile is None:
              raise IOError(errno.ENOENT, "File not found", path)
          if not isinstance(arvfile, ArvadosFile):
              raise IOError(errno.EISDIR, "Is a directory", path)
  
-        if mode[0] == "w":
+        if mode[0] == 'w':
              arvfile.truncate(0)
  
-        name = os.path.basename(path)
-
-        if mode == "r":
-            return ArvadosFileReader(arvfile, num_retries=self.num_retries)
-        else:
-            return ArvadosFileWriter(arvfile, mode, num_retries=self.num_retries)
+        return fclass(arvfile, mode=mode, num_retries=self.num_retries)
  
      def modified(self):
          """Determine if the collection has been modified since last commited."""
@@ -673,7 +682,7 @@ class RichCollectionBase(CollectionBase):
          if value == self._committed:
              return
          if value:
-            for k,v in self._items.items():
+            for k,v in listitems(self._items):
                  v.set_committed(True)
              self._committed = True
          else:
@@ -684,7 +693,7 @@ class RichCollectionBase(CollectionBase):
      @synchronized
      def __iter__(self):
          """Iterate over names of files and collections contained in this collection."""
-        return iter(self._items.keys())
+        return iter(viewkeys(self._items))
  
      @synchronized
      def __getitem__(self, k):
@@ -721,12 +730,12 @@ class RichCollectionBase(CollectionBase):
      @synchronized
      def values(self):
          """Get a list of files and collection objects directly contained in this collection."""
-        return self._items.values()
+        return listvalues(self._items)
  
      @synchronized
      def items(self):
          """Get a list of (name, object) tuples directly contained in this collection."""
-        return self._items.items()
+        return listitems(self._items)
  
      def exists(self, path):
          """Test if there is a file or collection at `path`."""
@@ -759,7 +768,7 @@ class RichCollectionBase(CollectionBase):
              item.remove(pathcomponents[1])
  
      def _clonefrom(self, source):
-        for k,v in source.items():
+        for k,v in listitems(source):
              self._items[k] = v.clone(self, k)
  
      def clone(self):
@@ -1075,8 +1084,8 @@ class RichCollectionBase(CollectionBase):
              # then return API server's PDH response.
              return self._portable_data_hash
          else:
-            stripped = self.portable_manifest_text()
-            return hashlib.md5(stripped).hexdigest() + '+' + str(len(stripped))
+            stripped = self.portable_manifest_text().encode()
+            return '{}+{}'.format(hashlib.md5(stripped).hexdigest(), len(stripped))
  
      @synchronized
      def subscribe(self, callback):
@@ -1117,7 +1126,7 @@ class RichCollectionBase(CollectionBase):
      @synchronized
      def flush(self):
          """Flush bufferblocks to Keep."""
-        for e in self.values():
+        for e in listvalues(self):
              e.flush()
  
  
@@ -1333,7 +1342,7 @@ class Collection(RichCollectionBase):
          # mode. Return an exception, or None if successful.
          try:
              self._manifest_text = self._my_keep().get(
-                self._manifest_locator, num_retries=self.num_retries)
+                self._manifest_locator, num_retries=self.num_retries).decode()
          except Exception as e:
              return e
  
@@ -1584,7 +1593,7 @@ class Collection(RichCollectionBase):
              if state == BLOCKS:
                  block_locator = re.match(r'[0-9a-f]{32}\+(\d+)(\+\S+)*', tok)
                  if block_locator:
-                    blocksize = long(block_locator.group(1))
+                    blocksize = int(block_locator.group(1))
                      blocks.append(Range(tok, streamoffset, blocksize, 0))
                      streamoffset += blocksize
                  else:
@@ -1593,8 +1602,8 @@ class Collection(RichCollectionBase):
              if state == SEGMENTS:
                  file_segment = re.search(r'^(\d+):(\d+):(\S+)', tok)
                  if file_segment:
-                    pos = long(file_segment.group(1))
-                    size = long(file_segment.group(2))
+                    pos = int(file_segment.group(1))
+                    size = int(file_segment.group(2))
                      name = file_segment.group(3).replace('\\040', ' ')
                      filepath = os.path.join(stream_name, name)
                      afile = self.find_or_create(filepath, FILE)