walk_object: rewrite as nonrecursive

[bup.git] / lib / bup / git.py
diff --git a/lib/bup/git.py b/lib/bup/git.py

index 82cd787c8ee7f148601747b673d4088bc5f6935e..460d6b11a2d9273aa0b2dfd4f0bfcb2599df5fe4 100644 (file)
--- a/lib/bup/git.py
+++ b/lib/bup/git.py
@@ -2,11 +2,19 @@
  bup repositories are in Git format. This library allows us to
  interact with the Git data structures.
  """
-import os, sys, zlib, time, subprocess, struct, stat, re, tempfile, glob
+
+import errno, os, sys, zlib, time, subprocess, struct, stat, re, tempfile, glob
  from collections import namedtuple
+from itertools import islice
+
+from bup import _helpers, hashsplit, path, midx, bloom, xstat
+from bup.helpers import (Sha1, add_error, chunkyreader, debug1, debug2,
+                         fdatasync,
+                         hostname, localtime, log, merge_iter,
+                         mmap_read, mmap_readwrite,
+                         progress, qprogress, unlink, username, userfullname,
+                         utc_offset_str)
  
-from bup.helpers import *
-from bup import _helpers, path, midx, bloom, xstat
  
  max_pack_size = 1000*1000*1000  # larger packs will slow down pruning
  max_pack_objects = 200*1000  # cache memory usage is about 83 bytes per object
@@ -86,18 +94,32 @@ def get_commit_items(id, cp):
      return parse_commit(commit_content)
  
  
-def repo(sub = ''):
+def _local_git_date_str(epoch_sec):
+    return '%d %s' % (epoch_sec, utc_offset_str(epoch_sec))
+
+
+def _git_date_str(epoch_sec, tz_offset_sec):
+    offs =  tz_offset_sec // 60
+    return '%d %s%02d%02d' \
+        % (epoch_sec,
+           '+' if offs >= 0 else '-',
+           abs(offs) // 60,
+           abs(offs) % 60)
+
+
+def repo(sub = '', repo_dir=None):
      """Get the path to the git repository or one of its subdirectories."""
      global repodir
-    if not repodir:
+    repo_dir = repo_dir or repodir
+    if not repo_dir:
          raise GitError('You should call check_repo_or_die()')
  
      # If there's a .git subdirectory, then the actual repo is in there.
-    gd = os.path.join(repodir, '.git')
+    gd = os.path.join(repo_dir, '.git')
      if os.path.exists(gd):
          repodir = gd
  
-    return os.path.join(repodir, sub)
+    return os.path.join(repo_dir, sub)
  
  
  def shorten_hash(s):
@@ -127,7 +149,7 @@ def auto_midx(objdir):
      args = [path.exe(), 'midx', '--auto', '--dir', objdir]
      try:
          rv = subprocess.call(args, stdout=open('/dev/null', 'w'))
-    except OSError, e:
+    except OSError as e:
          # make sure 'args' gets printed to help with debugging
          add_error('%r: exception: %s' % (args, e))
          raise
@@ -137,7 +159,7 @@ def auto_midx(objdir):
      args = [path.exe(), 'bloom', '--dir', objdir]
      try:
          rv = subprocess.call(args, stdout=open('/dev/null', 'w'))
-    except OSError, e:
+    except OSError as e:
          # make sure 'args' gets printed to help with debugging
          add_error('%r: exception: %s' % (args, e))
          raise
@@ -152,6 +174,7 @@ def mangle_name(name, mode, gitmode):
      disambiguate normal files from segmented ones.
      """
      if stat.S_ISREG(mode) and not stat.S_ISREG(gitmode):
+        assert(stat.S_ISDIR(gitmode))
          return name + '.bup'
      elif name.endswith('.bup') or name[:-1].endswith('.bup'):
          return name + '.bupl'
@@ -160,7 +183,7 @@ def mangle_name(name, mode, gitmode):
  
  
  (BUP_NORMAL, BUP_CHUNKED) = (0,1)
-def demangle_name(name):
+def demangle_name(name, mode):
      """Remove name mangling from a file name, if necessary.
  
      The return value is a tuple (demangled_filename,mode), where mode is one of
@@ -175,6 +198,9 @@ def demangle_name(name):
          return (name[:-5], BUP_NORMAL)
      elif name.endswith('.bup'):
          return (name[:-4], BUP_CHUNKED)
+    elif name.endswith('.bupm'):
+        return (name[:-5],
+                BUP_CHUNKED if stat.S_ISDIR(mode) else BUP_NORMAL)
      else:
          return (name, BUP_NORMAL)
  
@@ -226,6 +252,8 @@ def tree_decode(buf):
  
  
  def _encode_packobj(type, content, compression_level=1):
+    if compression_level not in (0, 1, 2, 3, 4, 5, 6, 7, 8, 9):
+        raise ValueError('invalid compression level %s' % compression_level)
      szout = ''
      sz = len(content)
      szbits = (sz & 0x0f) | (_typemap[type]<<4)
@@ -237,10 +265,6 @@ def _encode_packobj(type, content, compression_level=1):
              break
          szbits = sz & 0x7f
          sz >>= 7
-    if compression_level > 9:
-        compression_level = 9
-    elif compression_level < 0:
-        compression_level = 0
      z = zlib.compressobj(compression_level)
      yield szout
      yield z.compress(content)
@@ -496,7 +520,7 @@ class PackIdxList:
                  if not d.get(full):
                      try:
                          ix = open_idx(full)
-                    except GitError, e:
+                    except GitError as e:
                          add_error(e)
                          continue
                      d[full] = ix
@@ -553,25 +577,44 @@ def idxmerge(idxlist, final_progress=True):
  def _make_objcache():
      return PackIdxList(repo('objects/pack'))
  
+# bup-gc assumes that it can disable all PackWriter activities
+# (bloom/midx/cache) via the constructor and close() arguments.
+
  class PackWriter:
      """Writes Git objects inside a pack file."""
-    def __init__(self, objcache_maker=_make_objcache, compression_level=1):
+    def __init__(self, objcache_maker=_make_objcache, compression_level=1,
+                 run_midx=True, on_pack_finish=None):
+        self.file = None
+        self.parentfd = None
          self.count = 0
          self.outbytes = 0
          self.filename = None
-        self.file = None
          self.idx = None
          self.objcache_maker = objcache_maker
          self.objcache = None
          self.compression_level = compression_level
+        self.run_midx=run_midx
+        self.on_pack_finish = on_pack_finish
  
      def __del__(self):
          self.close()
  
      def _open(self):
          if not self.file:
-            (fd,name) = tempfile.mkstemp(suffix='.pack', dir=repo('objects'))
-            self.file = os.fdopen(fd, 'w+b')
+            objdir = dir=repo('objects')
+            fd, name = tempfile.mkstemp(suffix='.pack', dir=objdir)
+            try:
+                self.file = os.fdopen(fd, 'w+b')
+            except:
+                os.close(fd)
+                raise
+            try:
+                self.parentfd = os.open(objdir, os.O_RDONLY)
+            except:
+                f = self.file
+                self.file = None
+                f.close()
+                raise
              assert(name.endswith('.pack'))
              self.filename = name[:-5]
              self.file.write('PACK\0\0\0\2\0\0\0\0')
@@ -588,7 +631,7 @@ class PackWriter:
          oneblob = ''.join(datalist)
          try:
              f.write(oneblob)
-        except IOError, e:
+        except IOError as e:
              raise GitError, e, sys.exc_info()[2]
          nw = len(oneblob)
          crc = zlib.crc32(oneblob) & 0xffffffff
@@ -616,7 +659,7 @@ class PackWriter:
  
      def breakpoint(self):
          """Clear byte and object counts and return the last processed id."""
-        id = self._end()
+        id = self._end(self.run_midx)
          self.outbytes = self.count = 0
          return id
  
@@ -632,11 +675,15 @@ class PackWriter:
          self._require_objcache()
          return self.objcache.exists(id, want_source=want_source)
  
+    def write(self, sha, type, content):
+        """Write an object to the pack file.  Fails if sha exists()."""
+        self._write(sha, type, content)
+
      def maybe_write(self, type, content):
          """Write an object to the pack file if not present and return its id."""
          sha = calc_hash(type, content)
          if not self.exists(sha):
-            self._write(sha, type, content)
+            self.write(sha, type, content)
              self._require_objcache()
              self.objcache.add(sha)
          return sha
@@ -650,55 +697,71 @@ class PackWriter:
          content = tree_encode(shalist)
          return self.maybe_write('tree', content)
  
-    def _new_commit(self, tree, parent, author, adate, committer, cdate, msg):
+    def new_commit(self, tree, parent,
+                   author, adate_sec, adate_tz,
+                   committer, cdate_sec, cdate_tz,
+                   msg):
+        """Create a commit object in the pack.  The date_sec values must be
+        epoch-seconds, and if a tz is None, the local timezone is assumed."""
+        if adate_tz:
+            adate_str = _git_date_str(adate_sec, adate_tz)
+        else:
+            adate_str = _local_git_date_str(adate_sec)
+        if cdate_tz:
+            cdate_str = _git_date_str(cdate_sec, cdate_tz)
+        else:
+            cdate_str = _local_git_date_str(cdate_sec)
          l = []
          if tree: l.append('tree %s' % tree.encode('hex'))
          if parent: l.append('parent %s' % parent.encode('hex'))
-        if author: l.append('author %s %s' % (author, _git_date(adate)))
-        if committer: l.append('committer %s %s' % (committer, _git_date(cdate)))
+        if author: l.append('author %s %s' % (author, adate_str))
+        if committer: l.append('committer %s %s' % (committer, cdate_str))
          l.append('')
          l.append(msg)
          return self.maybe_write('commit', '\n'.join(l))
  
-    def new_commit(self, parent, tree, date, msg):
-        """Create a commit object in the pack."""
-        userline = '%s <%s@%s>' % (userfullname(), username(), hostname())
-        commit = self._new_commit(tree, parent,
-                                  userline, date, userline, date,
-                                  msg)
-        return commit
-
      def abort(self):
          """Remove the pack file from disk."""
          f = self.file
          if f:
-            self.idx = None
+            pfd = self.parentfd
              self.file = None
-            f.close()
-            os.unlink(self.filename + '.pack')
+            self.parentfd = None
+            self.idx = None
+            try:
+                try:
+                    os.unlink(self.filename + '.pack')
+                finally:
+                    f.close()
+            finally:
+                if pfd is not None:
+                    os.close(pfd)
  
      def _end(self, run_midx=True):
          f = self.file
          if not f: return None
          self.file = None
-        self.objcache = None
-        idx = self.idx
-        self.idx = None
+        try:
+            self.objcache = None
+            idx = self.idx
+            self.idx = None
  
-        # update object count
-        f.seek(8)
-        cp = struct.pack('!i', self.count)
-        assert(len(cp) == 4)
-        f.write(cp)
-
-        # calculate the pack sha1sum
-        f.seek(0)
-        sum = Sha1()
-        for b in chunkyreader(f):
-            sum.update(b)
-        packbin = sum.digest()
-        f.write(packbin)
-        f.close()
+            # update object count
+            f.seek(8)
+            cp = struct.pack('!i', self.count)
+            assert(len(cp) == 4)
+            f.write(cp)
+
+            # calculate the pack sha1sum
+            f.seek(0)
+            sum = Sha1()
+            for b in chunkyreader(f):
+                sum.update(b)
+            packbin = sum.digest()
+            f.write(packbin)
+            fdatasync(f.fileno())
+        finally:
+            f.close()
  
          obj_list_sha = self._write_pack_idx_v2(self.filename + '.idx', idx, packbin)
  
@@ -707,9 +770,17 @@ class PackWriter:
              os.unlink(self.filename + '.map')
          os.rename(self.filename + '.pack', nameprefix + '.pack')
          os.rename(self.filename + '.idx', nameprefix + '.idx')
+        try:
+            os.fsync(self.parentfd)
+        finally:
+            os.close(self.parentfd)
  
          if run_midx:
              auto_midx(repo('objects/pack'))
+
+        if self.on_pack_finish:
+            self.on_pack_finish(nameprefix)
+
          return nameprefix
  
      def close(self, run_midx=True):
@@ -729,11 +800,15 @@ class PackWriter:
          idx_f = open(filename, 'w+b')
          try:
              idx_f.truncate(index_len)
+            fdatasync(idx_f.fileno())
              idx_map = mmap_readwrite(idx_f, close=False)
-            count = _helpers.write_idx(filename, idx_map, idx, self.count)
-            assert(count == self.count)
+            try:
+                count = _helpers.write_idx(filename, idx_map, idx, self.count)
+                assert(count == self.count)
+                idx_map.flush()
+            finally:
+                idx_map.close()
          finally:
-            if idx_map: idx_map.close()
              idx_f.close()
  
          idx_f = open(filename, 'a+b')
@@ -753,27 +828,40 @@ class PackWriter:
              for b in chunkyreader(idx_f):
                  idx_sum.update(b)
              idx_f.write(idx_sum.digest())
+            fdatasync(idx_f.fileno())
              return namebase
          finally:
              idx_f.close()
  
  
-def _git_date(date):
-    return '%d %s' % (date, time.strftime('%z', time.localtime(date)))
+def _gitenv(repo_dir = None):
+    if not repo_dir:
+        repo_dir = repo()
+    def env():
+        os.environ['GIT_DIR'] = os.path.abspath(repo_dir)
+    return env
  
  
-def _gitenv():
-    os.environ['GIT_DIR'] = os.path.abspath(repo())
+def list_refs(refname=None, repo_dir=None,
+              limit_to_heads=False, limit_to_tags=False):
+    """Yield (refname, hash) tuples for all repository refs unless a ref
+    name is specified.  Given a ref name, only include tuples for that
+    particular ref.  The limits restrict the result items to
+    refs/heads or refs/tags.  If both limits are specified, items from
+    both sources will be included.
  
-
-def list_refs(refname = None):
-    """Generate a list of tuples in the form (refname,hash).
-    If a ref name is specified, list only this particular ref.
      """
-    argv = ['git', 'show-ref', '--']
+    argv = ['git', 'show-ref']
+    if limit_to_heads:
+        argv.append('--heads')
+    if limit_to_tags:
+        argv.append('--tags')
+    argv.append('--')
      if refname:
          argv += [refname]
-    p = subprocess.Popen(argv, preexec_fn = _gitenv, stdout = subprocess.PIPE)
+    p = subprocess.Popen(argv,
+                         preexec_fn = _gitenv(repo_dir),
+                         stdout = subprocess.PIPE)
      out = p.stdout.read().strip()
      rv = p.wait()  # not fatal
      if rv:
@@ -784,9 +872,10 @@ def list_refs(refname = None):
              yield (name, sha.decode('hex'))
  
  
-def read_ref(refname):
+def read_ref(refname, repo_dir = None):
      """Get the commit id of the most recent commit made on a given ref."""
-    l = list(list_refs(refname))
+    refs = list_refs(refname, repo_dir=repo_dir, limit_to_heads=True)
+    l = tuple(islice(refs, 2))
      if l:
          assert(len(l) == 1)
          return l[0][1]
@@ -794,7 +883,7 @@ def read_ref(refname):
          return None
  
  
-def rev_list(ref, count=None):
+def rev_list(ref, count=None, repo_dir=None):
      """Generate a list of reachable commits in reverse chronological order.
  
      This generator walks through commits, from child to parent, that are
@@ -809,7 +898,9 @@ def rev_list(ref, count=None):
      if count:
          opts += ['-n', str(atoi(count))]
      argv = ['git', 'rev-list', '--pretty=format:%at'] + opts + [ref, '--']
-    p = subprocess.Popen(argv, preexec_fn = _gitenv, stdout = subprocess.PIPE)
+    p = subprocess.Popen(argv,
+                         preexec_fn = _gitenv(repo_dir),
+                         stdout = subprocess.PIPE)
      commit = None
      for row in p.stdout:
          s = row.strip()
@@ -823,18 +914,18 @@ def rev_list(ref, count=None):
          raise GitError, 'git rev-list returned error %d' % rv
  
  
-def get_commit_dates(refs):
+def get_commit_dates(refs, repo_dir=None):
      """Get the dates for the specified commit refs.  For now, every unique
         string in refs must resolve to a different commit or this
         function will fail."""
      result = []
      for ref in refs:
-        commit = get_commit_items(ref, cp())
+        commit = get_commit_items(ref, cp(repo_dir))
          result.append(commit.author_sec)
      return result
  
  
-def rev_parse(committish):
+def rev_parse(committish, repo_dir=None):
      """Resolve the full hash for 'committish', if it exists.
  
      Should be roughly equivalent to 'git rev-parse'.
@@ -842,12 +933,12 @@ def rev_parse(committish):
      Returns the hex value of the hash if it is found, None if 'committish' does
      not correspond to anything.
      """
-    head = read_ref(committish)
+    head = read_ref(committish, repo_dir=repo_dir)
      if head:
          debug2("resolved from ref: commit = %s\n" % head.encode('hex'))
          return head
  
-    pL = PackIdxList(repo('objects/pack'))
+    pL = PackIdxList(repo('objects/pack', repo_dir=repo_dir))
  
      if len(committish) == 40:
          try:
@@ -861,14 +952,24 @@ def rev_parse(committish):
      return None
  
  
-def update_ref(refname, newval, oldval):
-    """Change the commit pointed to by a branch."""
+def update_ref(refname, newval, oldval, repo_dir=None):
+    """Update a repository reference."""
      if not oldval:
          oldval = ''
-    assert(refname.startswith('refs/heads/'))
+    assert(refname.startswith('refs/heads/') \
+           or refname.startswith('refs/tags/'))
      p = subprocess.Popen(['git', 'update-ref', refname,
                            newval.encode('hex'), oldval.encode('hex')],
-                         preexec_fn = _gitenv)
+                         preexec_fn = _gitenv(repo_dir))
+    _git_wait('git update-ref', p)
+
+
+def delete_ref(refname, oldvalue=None):
+    """Delete a repository reference (see git update-ref(1))."""
+    assert(refname.startswith('refs/'))
+    oldvalue = [] if not oldvalue else [oldvalue]
+    p = subprocess.Popen(['git', 'update-ref', '-d', refname] + oldvalue,
+                         preexec_fn = _gitenv())
      _git_wait('git update-ref', p)
  
  
@@ -898,16 +999,16 @@ def init_repo(path=None):
      if os.path.exists(d) and not os.path.isdir(os.path.join(d, '.')):
          raise GitError('"%s" exists but is not a directory\n' % d)
      p = subprocess.Popen(['git', '--bare', 'init'], stdout=sys.stderr,
-                         preexec_fn = _gitenv)
+                         preexec_fn = _gitenv())
      _git_wait('git init', p)
      # Force the index version configuration in order to ensure bup works
      # regardless of the version of the installed Git binary.
      p = subprocess.Popen(['git', 'config', 'pack.indexVersion', '2'],
-                         stdout=sys.stderr, preexec_fn = _gitenv)
+                         stdout=sys.stderr, preexec_fn = _gitenv())
      _git_wait('git config', p)
      # Enable the reflog
      p = subprocess.Popen(['git', 'config', 'core.logAllRefUpdates', 'true'],
-                         stdout=sys.stderr, preexec_fn = _gitenv)
+                         stdout=sys.stderr, preexec_fn = _gitenv())
      _git_wait('git config', p)
  
  
@@ -919,7 +1020,7 @@ def check_repo_or_die(path=None):
      guess_repo(path)
      try:
          os.stat(repo('objects/pack/.'))
-    except OSError, e:
+    except OSError as e:
          if e.errno == errno.ENOENT:
              log('error: %r is not a bup repository; run "bup init"\n'
                  % repo())
@@ -963,7 +1064,7 @@ def _git_wait(cmd, p):
  
  
  def _git_capture(argv):
-    p = subprocess.Popen(argv, stdout=subprocess.PIPE, preexec_fn = _gitenv)
+    p = subprocess.Popen(argv, stdout=subprocess.PIPE, preexec_fn = _gitenv())
      r = p.stdout.read()
      _git_wait(repr(argv), p)
      return r
@@ -981,7 +1082,7 @@ class _AbortableIter:
      def next(self):
          try:
              return self.it.next()
-        except StopIteration, e:
+        except StopIteration as e:
              self.done = True
              raise
          except:
@@ -999,11 +1100,18 @@ class _AbortableIter:
          self.abort()
  
  
+class MissingObject(KeyError):
+    def __init__(self, id):
+        self.id = id
+        KeyError.__init__(self, 'object %r is missing' % id.encode('hex'))
+
+
  _ver_warned = 0
  class CatPipe:
      """Link to 'git cat-file' that is used to retrieve blob data."""
-    def __init__(self):
+    def __init__(self, repo_dir = None):
          global _ver_warned
+        self.repo_dir = repo_dir
          wanted = ('1','5','6')
          if ver() < wanted:
              if not _ver_warned:
@@ -1029,7 +1137,7 @@ class CatPipe:
                                    stdout=subprocess.PIPE,
                                    close_fds = True,
                                    bufsize = 4096,
-                                  preexec_fn = _gitenv)
+                                  preexec_fn = _gitenv(self.repo_dir))
  
      def _fast_get(self, id):
          if not self.p or self.p.poll() != None:
@@ -1050,7 +1158,7 @@ class CatPipe:
          hdr = self.p.stdout.readline()
          if hdr.endswith(' missing\n'):
              self.inprogress = None
-            raise KeyError('blob %r is missing' % id)
+            raise MissingObject(id.decode('hex'))
          spl = hdr.split(' ')
          if len(spl) != 3 or len(spl[0]) != 40:
              raise GitError('expected blob, got %r' % spl)
@@ -1065,7 +1173,7 @@ class CatPipe:
              readline_result = self.p.stdout.readline()
              assert(readline_result == '\n')
              self.inprogress = None
-        except Exception, e:
+        except Exception as e:
              it.abort()
              raise
  
@@ -1078,7 +1186,7 @@ class CatPipe:
  
          p = subprocess.Popen(['git', 'cat-file', type, id],
                               stdout=subprocess.PIPE,
-                             preexec_fn = _gitenv)
+                             preexec_fn = _gitenv(self.repo_dir))
          for blob in chunkyreader(p.stdout):
              yield blob
          _git_wait('git cat-file', p)
@@ -1115,28 +1223,98 @@ class CatPipe:
              log('booger!\n')
  
  
-_cp = (None, None)
+_cp = {}
  
-def cp():
-    """Create a CatPipe object or reuse an already existing one."""
+def cp(repo_dir=None):
+    """Create a CatPipe object or reuse the already existing one."""
      global _cp
-    cp_dir, cp = _cp
-    cur_dir = os.path.realpath(repo())
-    if cur_dir != cp_dir:
-        cp = CatPipe()
-        _cp = (cur_dir, cp)
+    if not repo_dir:
+        repo_dir = repo()
+    repo_dir = os.path.abspath(repo_dir)
+    cp = _cp.get(repo_dir)
+    if not cp:
+        cp = CatPipe(repo_dir)
+        _cp[repo_dir] = cp
      return cp
  
  
-def tags():
+def tags(repo_dir = None):
      """Return a dictionary of all tags in the form {hash: [tag_names, ...]}."""
      tags = {}
-    for (n,c) in list_refs():
-        if n.startswith('refs/tags/'):
-            name = n[10:]
-            if not c in tags:
-                tags[c] = []
+    for n, c in list_refs(repo_dir = repo_dir, limit_to_tags=True):
+        assert(n.startswith('refs/tags/'))
+        name = n[10:]
+        if not c in tags:
+            tags[c] = []
+        tags[c].append(name)  # more than one tag can point at 'c'
+    return tags
  
-            tags[c].append(name)  # more than one tag can point at 'c'
  
-    return tags
+WalkItem = namedtuple('WalkItem', ['id', 'type', 'mode',
+                                   'path', 'chunk_path', 'data'])
+# The path is the mangled path, and if an item represents a fragment
+# of a chunked file, the chunk_path will be the chunked subtree path
+# for the chunk, i.e. ['', '2d3115e', ...].  The top-level path for a
+# chunked file will have a chunk_path of [''].  So some chunk subtree
+# of the file '/foo/bar/baz' might look like this:
+#
+#   item.path = ['foo', 'bar', 'baz.bup']
+#   item.chunk_path = ['', '2d3115e', '016b097']
+#   item.type = 'tree'
+#   ...
+
+
+def walk_object(cat_pipe, id,
+                stop_at=None,
+                include_data=None):
+    """Yield everything reachable from id via cat_pipe as a WalkItem,
+    stopping whenever stop_at(id) returns true.  Throw MissingObject
+    if a hash encountered is missing from the repository.
+
+    """
+    # Maintain the pending stack on the heap to avoid stack overflow
+    pending = [(id, [], [], None)]
+    while len(pending):
+        id, parent_path, chunk_path, mode = pending.pop()
+        if stop_at and stop_at(id):
+            continue
+
+        item_it = cat_pipe.get(id)  # FIXME: use include_data
+        type = item_it.next()
+        if type not in ('blob', 'commit', 'tree'):
+            raise Exception('unexpected repository object type %r' % type)
+
+        # FIXME: set the mode based on the type when the mode is None
+        if type == 'blob' and not include_data:
+            # Dump data until we can ask cat_pipe not to fetch it
+            for ignored in item_it:
+                pass
+            data = None
+        else:
+            data = ''.join(item_it)
+
+        yield WalkItem(id=id, type=type,
+                       chunk_path=chunk_path, path=parent_path,
+                       mode=mode,
+                       data=(data if include_data else None))
+
+        if type == 'commit':
+            commit_items = parse_commit(data)
+            for pid in commit_items.parents:
+                pending.append((pid, parent_path, chunk_path, mode))
+            pending.append((commit_items.tree, parent_path, chunk_path,
+                            hashsplit.GIT_MODE_TREE))
+        elif type == 'tree':
+            for mode, name, ent_id in tree_decode(data):
+                demangled, bup_type = demangle_name(name, mode)
+                if chunk_path:
+                    sub_path = parent_path
+                    sub_chunk_path = chunk_path + [name]
+                else:
+                    sub_path = parent_path + [name]
+                    if bup_type == BUP_CHUNKED:
+                        sub_chunk_path = ['']
+                    else:
+                        sub_chunk_path = chunk_path
+                pending.append((ent_id.encode('hex'), sub_path, sub_chunk_path,
+                                mode))