Honor git config pack.packSizeLimit when set

[bup.git] / lib / bup / git.py
diff --git a/lib/bup/git.py b/lib/bup/git.py

index 6db392a2226fe5de78df7d4c15c44982a18bfdd2..950a2d4ebb41e0399851d8b71ee49d6e9c301d5b 100644 (file)
--- a/lib/bup/git.py
+++ b/lib/bup/git.py
@@ -2,16 +2,24 @@
  bup repositories are in Git format. This library allows us to
  interact with the Git data structures.
  """
  bup repositories are in Git format. This library allows us to
  interact with the Git data structures.
  """
-import os, sys, zlib, time, subprocess, struct, stat, re, tempfile, glob
-from bup.helpers import *
-from bup import _helpers, path, midx, bloom
  
  
-SEEK_END=2  # os.SEEK_END is not defined in python 2.4
+import errno, os, sys, zlib, time, subprocess, struct, stat, re, tempfile, glob
+from collections import namedtuple
+from itertools import islice
+
+from bup import _helpers, hashsplit, path, midx, bloom, xstat
+from bup.helpers import (Sha1, add_error, chunkyreader, debug1, debug2,
+                         fdatasync,
+                         hostname, localtime, log, merge_iter,
+                         mmap_read, mmap_readwrite,
+                         parse_num,
+                         progress, qprogress, stat_if_exists,
+                         unlink, username, userfullname,
+                         utc_offset_str)
  
  verbose = 0
  ignore_midx = 0
  
  verbose = 0
  ignore_midx = 0
-home_repodir = os.path.expanduser('~/.bup')
-repodir = None
+repodir = None  # The default repository, once initialized
  
  _typemap =  { 'blob':3, 'tree':2, 'commit':1, 'tag':4 }
  _typermap = { 3:'blob', 2:'tree', 1:'commit', 4:'tag' }
  
  _typemap =  { 'blob':3, 'tree':2, 'commit':1, 'tag':4 }
  _typermap = { 3:'blob', 2:'tree', 1:'commit', 4:'tag' }
@@ -24,18 +32,116 @@ class GitError(Exception):
      pass
  
  
      pass
  
  
-def repo(sub = ''):
+def _git_wait(cmd, p):
+    rv = p.wait()
+    if rv != 0:
+        raise GitError('%s returned %d' % (cmd, rv))
+
+def _git_capture(argv):
+    p = subprocess.Popen(argv, stdout=subprocess.PIPE, preexec_fn = _gitenv())
+    r = p.stdout.read()
+    _git_wait(repr(argv), p)
+    return r
+
+def git_config_get(option, repo_dir=None):
+    cmd = ('git', 'config', '--get', option)
+    p = subprocess.Popen(cmd, stdout=subprocess.PIPE,
+                         preexec_fn=_gitenv(repo_dir=repo_dir))
+    r = p.stdout.read()
+    rc = p.wait()
+    if rc == 0:
+        return r
+    if rc != 1:
+        raise GitError('%s returned %d' % (cmd, rc))
+    return None
+
+
+def parse_tz_offset(s):
+    """UTC offset in seconds."""
+    tz_off = (int(s[1:3]) * 60 * 60) + (int(s[3:5]) * 60)
+    if s[0] == '-':
+        return - tz_off
+    return tz_off
+
+
+# FIXME: derived from http://git.rsbx.net/Documents/Git_Data_Formats.txt
+# Make sure that's authoritative.
+_start_end_char = r'[^ .,:;<>"\'\0\n]'
+_content_char = r'[^\0\n<>]'
+_safe_str_rx = '(?:%s{1,2}|(?:%s%s*%s))' \
+    % (_start_end_char,
+       _start_end_char, _content_char, _start_end_char)
+_tz_rx = r'[-+]\d\d[0-5]\d'
+_parent_rx = r'(?:parent [abcdefABCDEF0123456789]{40}\n)'
+_commit_rx = re.compile(r'''tree (?P<tree>[abcdefABCDEF0123456789]{40})
+(?P<parents>%s*)author (?P<author_name>%s) <(?P<author_mail>%s)> (?P<asec>\d+) (?P<atz>%s)
+committer (?P<committer_name>%s) <(?P<committer_mail>%s)> (?P<csec>\d+) (?P<ctz>%s)
+
+(?P<message>(?:.|\n)*)''' % (_parent_rx,
+                             _safe_str_rx, _safe_str_rx, _tz_rx,
+                             _safe_str_rx, _safe_str_rx, _tz_rx))
+_parent_hash_rx = re.compile(r'\s*parent ([abcdefABCDEF0123456789]{40})\s*')
+
+
+# Note that the author_sec and committer_sec values are (UTC) epoch seconds.
+CommitInfo = namedtuple('CommitInfo', ['tree', 'parents',
+                                       'author_name', 'author_mail',
+                                       'author_sec', 'author_offset',
+                                       'committer_name', 'committer_mail',
+                                       'committer_sec', 'committer_offset',
+                                       'message'])
+
+def parse_commit(content):
+    commit_match = re.match(_commit_rx, content)
+    if not commit_match:
+        raise Exception('cannot parse commit %r' % content)
+    matches = commit_match.groupdict()
+    return CommitInfo(tree=matches['tree'],
+                      parents=re.findall(_parent_hash_rx, matches['parents']),
+                      author_name=matches['author_name'],
+                      author_mail=matches['author_mail'],
+                      author_sec=int(matches['asec']),
+                      author_offset=parse_tz_offset(matches['atz']),
+                      committer_name=matches['committer_name'],
+                      committer_mail=matches['committer_mail'],
+                      committer_sec=int(matches['csec']),
+                      committer_offset=parse_tz_offset(matches['ctz']),
+                      message=matches['message'])
+
+
+def get_commit_items(id, cp):
+    commit_it = cp.get(id)
+    assert(commit_it.next() == 'commit')
+    commit_content = ''.join(commit_it)
+    return parse_commit(commit_content)
+
+
+def _local_git_date_str(epoch_sec):
+    return '%d %s' % (epoch_sec, utc_offset_str(epoch_sec))
+
+
+def _git_date_str(epoch_sec, tz_offset_sec):
+    offs =  tz_offset_sec // 60
+    return '%d %s%02d%02d' \
+        % (epoch_sec,
+           '+' if offs >= 0 else '-',
+           abs(offs) // 60,
+           abs(offs) % 60)
+
+
+def repo(sub = '', repo_dir=None):
      """Get the path to the git repository or one of its subdirectories."""
      global repodir
      """Get the path to the git repository or one of its subdirectories."""
      global repodir
-    if not repodir:
+    repo_dir = repo_dir or repodir
+    if not repo_dir:
          raise GitError('You should call check_repo_or_die()')
  
      # If there's a .git subdirectory, then the actual repo is in there.
          raise GitError('You should call check_repo_or_die()')
  
      # If there's a .git subdirectory, then the actual repo is in there.
-    gd = os.path.join(repodir, '.git')
+    gd = os.path.join(repo_dir, '.git')
      if os.path.exists(gd):
          repodir = gd
  
      if os.path.exists(gd):
          repodir = gd
  
-    return os.path.join(repodir, sub)
+    return os.path.join(repo_dir, sub)
  
  
  def shorten_hash(s):
  
  
  def shorten_hash(s):
@@ -65,7 +171,7 @@ def auto_midx(objdir):
      args = [path.exe(), 'midx', '--auto', '--dir', objdir]
      try:
          rv = subprocess.call(args, stdout=open('/dev/null', 'w'))
      args = [path.exe(), 'midx', '--auto', '--dir', objdir]
      try:
          rv = subprocess.call(args, stdout=open('/dev/null', 'w'))
-    except OSError, e:
+    except OSError as e:
          # make sure 'args' gets printed to help with debugging
          add_error('%r: exception: %s' % (args, e))
          raise
          # make sure 'args' gets printed to help with debugging
          add_error('%r: exception: %s' % (args, e))
          raise
@@ -75,7 +181,7 @@ def auto_midx(objdir):
      args = [path.exe(), 'bloom', '--dir', objdir]
      try:
          rv = subprocess.call(args, stdout=open('/dev/null', 'w'))
      args = [path.exe(), 'bloom', '--dir', objdir]
      try:
          rv = subprocess.call(args, stdout=open('/dev/null', 'w'))
-    except OSError, e:
+    except OSError as e:
          # make sure 'args' gets printed to help with debugging
          add_error('%r: exception: %s' % (args, e))
          raise
          # make sure 'args' gets printed to help with debugging
          add_error('%r: exception: %s' % (args, e))
          raise
@@ -87,9 +193,10 @@ def mangle_name(name, mode, gitmode):
      """Mangle a file name to present an abstract name for segmented files.
      Mangled file names will have the ".bup" extension added to them. If a
      file's name already ends with ".bup", a ".bupl" extension is added to
      """Mangle a file name to present an abstract name for segmented files.
      Mangled file names will have the ".bup" extension added to them. If a
      file's name already ends with ".bup", a ".bupl" extension is added to
-    disambiguate normal files from semgmented ones.
+    disambiguate normal files from segmented ones.
      """
      if stat.S_ISREG(mode) and not stat.S_ISREG(gitmode):
      """
      if stat.S_ISREG(mode) and not stat.S_ISREG(gitmode):
+        assert(stat.S_ISDIR(gitmode))
          return name + '.bup'
      elif name.endswith('.bup') or name[:-1].endswith('.bup'):
          return name + '.bupl'
          return name + '.bup'
      elif name.endswith('.bup') or name[:-1].endswith('.bup'):
          return name + '.bupl'
@@ -98,26 +205,77 @@ def mangle_name(name, mode, gitmode):
  
  
  (BUP_NORMAL, BUP_CHUNKED) = (0,1)
  
  
  (BUP_NORMAL, BUP_CHUNKED) = (0,1)
-def demangle_name(name):
+def demangle_name(name, mode):
      """Remove name mangling from a file name, if necessary.
  
      The return value is a tuple (demangled_filename,mode), where mode is one of
      the following:
  
      * BUP_NORMAL  : files that should be read as-is from the repository
      """Remove name mangling from a file name, if necessary.
  
      The return value is a tuple (demangled_filename,mode), where mode is one of
      the following:
  
      * BUP_NORMAL  : files that should be read as-is from the repository
-    * BUP_CHUNKED : files that were chunked and need to be assembled
+    * BUP_CHUNKED : files that were chunked and need to be reassembled
  
  
-    For more information on the name mangling algorythm, see mangle_name()
+    For more information on the name mangling algorithm, see mangle_name()
      """
      if name.endswith('.bupl'):
          return (name[:-5], BUP_NORMAL)
      elif name.endswith('.bup'):
          return (name[:-4], BUP_CHUNKED)
      """
      if name.endswith('.bupl'):
          return (name[:-5], BUP_NORMAL)
      elif name.endswith('.bup'):
          return (name[:-4], BUP_CHUNKED)
+    elif name.endswith('.bupm'):
+        return (name[:-5],
+                BUP_CHUNKED if stat.S_ISDIR(mode) else BUP_NORMAL)
      else:
          return (name, BUP_NORMAL)
  
  
      else:
          return (name, BUP_NORMAL)
  
  
-def _encode_packobj(type, content):
+def calc_hash(type, content):
+    """Calculate some content's hash in the Git fashion."""
+    header = '%s %d\0' % (type, len(content))
+    sum = Sha1(header)
+    sum.update(content)
+    return sum.digest()
+
+
+def shalist_item_sort_key(ent):
+    (mode, name, id) = ent
+    assert(mode+0 == mode)
+    if stat.S_ISDIR(mode):
+        return name + '/'
+    else:
+        return name
+
+
+def tree_encode(shalist):
+    """Generate a git tree object from (mode,name,hash) tuples."""
+    shalist = sorted(shalist, key = shalist_item_sort_key)
+    l = []
+    for (mode,name,bin) in shalist:
+        assert(mode)
+        assert(mode+0 == mode)
+        assert(name)
+        assert(len(bin) == 20)
+        s = '%o %s\0%s' % (mode,name,bin)
+        assert(s[0] != '0')  # 0-padded octal is not acceptable in a git tree
+        l.append(s)
+    return ''.join(l)
+
+
+def tree_decode(buf):
+    """Generate a list of (mode,name,hash) from the git tree object in buf."""
+    ofs = 0
+    while ofs < len(buf):
+        z = buf.find('\0', ofs)
+        assert(z > ofs)
+        spl = buf[ofs:z].split(' ', 1)
+        assert(len(spl) == 2)
+        mode,name = spl
+        sha = buf[z+1:z+1+20]
+        ofs = z+1+20
+        yield (int(mode, 8), name, sha)
+
+
+def _encode_packobj(type, content, compression_level=1):
+    if compression_level not in (0, 1, 2, 3, 4, 5, 6, 7, 8, 9):
+        raise ValueError('invalid compression level %s' % compression_level)
      szout = ''
      sz = len(content)
      szbits = (sz & 0x0f) | (_typemap[type]<<4)
      szout = ''
      sz = len(content)
      szbits = (sz & 0x0f) | (_typemap[type]<<4)
@@ -129,14 +287,14 @@ def _encode_packobj(type, content):
              break
          szbits = sz & 0x7f
          sz >>= 7
              break
          szbits = sz & 0x7f
          sz >>= 7
-    z = zlib.compressobj(1)
+    z = zlib.compressobj(compression_level)
      yield szout
      yield z.compress(content)
      yield z.flush()
  
  
      yield szout
      yield z.compress(content)
      yield z.flush()
  
  
-def _encode_looseobj(type, content):
-    z = zlib.compressobj(1)
+def _encode_looseobj(type, content, compression_level=1):
+    z = zlib.compressobj(compression_level)
      yield z.compress('%s %d\0' % (type, len(content)))
      yield z.compress(content)
      yield z.flush()
      yield z.compress('%s %d\0' % (type, len(content)))
      yield z.compress(content)
      yield z.flush()
@@ -356,12 +514,13 @@ class PackIdxList:
                                      '  used by %s\n') % (n, mxf))
                                  broken = True
                          if broken:
                                      '  used by %s\n') % (n, mxf))
                                  broken = True
                          if broken:
+                            mx.close()
                              del mx
                              unlink(full)
                          else:
                              midxl.append(mx)
                  midxl.sort(key=lambda ix:
                              del mx
                              unlink(full)
                          else:
                              midxl.append(mx)
                  midxl.sort(key=lambda ix:
-                           (-len(ix), -os.stat(ix.name).st_mtime))
+                           (-len(ix), -xstat.stat(ix.name).st_mtime))
                  for ix in midxl:
                      any_needed = False
                      for sub in ix.idxnames:
                  for ix in midxl:
                      any_needed = False
                      for sub in ix.idxnames:
@@ -377,12 +536,13 @@ class PackIdxList:
                      elif not ix.force_keep:
                          debug1('midx: removing redundant: %s\n'
                                 % os.path.basename(ix.name))
                      elif not ix.force_keep:
                          debug1('midx: removing redundant: %s\n'
                                 % os.path.basename(ix.name))
+                        ix.close()
                          unlink(ix.name)
              for full in glob.glob(os.path.join(self.dir,'*.idx')):
                  if not d.get(full):
                      try:
                          ix = open_idx(full)
                          unlink(ix.name)
              for full in glob.glob(os.path.join(self.dir,'*.idx')):
                  if not d.get(full):
                      try:
                          ix = open_idx(full)
-                    except GitError, e:
+                    except GitError as e:
                          add_error(e)
                          continue
                      d[full] = ix
                          add_error(e)
                          continue
                      d[full] = ix
@@ -403,22 +563,6 @@ class PackIdxList:
          self.also.add(hash)
  
  
          self.also.add(hash)
  
  
-def calc_hash(type, content):
-    """Calculate some content's hash in the Git fashion."""
-    header = '%s %d\0' % (type, len(content))
-    sum = Sha1(header)
-    sum.update(content)
-    return sum.digest()
-
-
-def _shalist_sort_key(ent):
-    (mode, name, id) = ent
-    if stat.S_ISDIR(int(mode, 8)):
-        return name + '/'
-    else:
-        return name
-
-
  def open_idx(filename):
      if filename.endswith('.idx'):
          f = open(filename, 'rb')
  def open_idx(filename):
      if filename.endswith('.idx'):
          f = open(filename, 'rb')
@@ -455,24 +599,58 @@ def idxmerge(idxlist, final_progress=True):
  def _make_objcache():
      return PackIdxList(repo('objects/pack'))
  
  def _make_objcache():
      return PackIdxList(repo('objects/pack'))
  
+# bup-gc assumes that it can disable all PackWriter activities
+# (bloom/midx/cache) via the constructor and close() arguments.
+
  class PackWriter:
      """Writes Git objects inside a pack file."""
  class PackWriter:
      """Writes Git objects inside a pack file."""
-    def __init__(self, objcache_maker=_make_objcache):
+    def __init__(self, objcache_maker=_make_objcache, compression_level=1,
+                 run_midx=True, on_pack_finish=None,
+                 max_pack_size=None, max_pack_objects=None):
+        self.repo_dir = repo()
+        self.file = None
+        self.parentfd = None
          self.count = 0
          self.outbytes = 0
          self.filename = None
          self.count = 0
          self.outbytes = 0
          self.filename = None
-        self.file = None
          self.idx = None
          self.objcache_maker = objcache_maker
          self.objcache = None
          self.idx = None
          self.objcache_maker = objcache_maker
          self.objcache = None
+        self.compression_level = compression_level
+        self.run_midx=run_midx
+        self.on_pack_finish = on_pack_finish
+        if not max_pack_size:
+            max_pack_size = git_config_get('pack.packSizeLimit',
+                                           repo_dir=self.repo_dir)
+            if max_pack_size is not None:
+                max_pack_size = parse_num(max_pack_size)
+            if not max_pack_size:
+                # larger packs slow down pruning
+                max_pack_size = 1000 * 1000 * 1000
+        self.max_pack_size = max_pack_size
+        # cache memory usage is about 83 bytes per object
+        self.max_pack_objects = max_pack_objects if max_pack_objects \
+                                else max(1, self.max_pack_size // 5000)
  
      def __del__(self):
          self.close()
  
      def _open(self):
          if not self.file:
  
      def __del__(self):
          self.close()
  
      def _open(self):
          if not self.file:
-            (fd,name) = tempfile.mkstemp(suffix='.pack', dir=repo('objects'))
-            self.file = os.fdopen(fd, 'w+b')
+            objdir = dir = os.path.join(self.repo_dir, 'objects')
+            fd, name = tempfile.mkstemp(suffix='.pack', dir=objdir)
+            try:
+                self.file = os.fdopen(fd, 'w+b')
+            except:
+                os.close(fd)
+                raise
+            try:
+                self.parentfd = os.open(objdir, os.O_RDONLY)
+            except:
+                f = self.file
+                self.file = None
+                f.close()
+                raise
              assert(name.endswith('.pack'))
              self.filename = name[:-5]
              self.file.write('PACK\0\0\0\2\0\0\0\0')
              assert(name.endswith('.pack'))
              self.filename = name[:-5]
              self.file.write('PACK\0\0\0\2\0\0\0\0')
@@ -489,7 +667,7 @@ class PackWriter:
          oneblob = ''.join(datalist)
          try:
              f.write(oneblob)
          oneblob = ''.join(datalist)
          try:
              f.write(oneblob)
-        except IOError, e:
+        except IOError as e:
              raise GitError, e, sys.exc_info()[2]
          nw = len(oneblob)
          crc = zlib.crc32(oneblob) & 0xffffffff
              raise GitError, e, sys.exc_info()[2]
          nw = len(oneblob)
          crc = zlib.crc32(oneblob) & 0xffffffff
@@ -508,12 +686,17 @@ class PackWriter:
              log('>')
          if not sha:
              sha = calc_hash(type, content)
              log('>')
          if not sha:
              sha = calc_hash(type, content)
-        size, crc = self._raw_write(_encode_packobj(type, content), sha=sha)
+        size, crc = self._raw_write(_encode_packobj(type, content,
+                                                    self.compression_level),
+                                    sha=sha)
+        if self.outbytes >= self.max_pack_size \
+           or self.count >= self.max_pack_objects:
+            self.breakpoint()
          return sha
  
      def breakpoint(self):
          """Clear byte and object counts and return the last processed id."""
          return sha
  
      def breakpoint(self):
          """Clear byte and object counts and return the last processed id."""
-        id = self._end()
+        id = self._end(self.run_midx)
          self.outbytes = self.count = 0
          return id
  
          self.outbytes = self.count = 0
          return id
  
@@ -529,12 +712,17 @@ class PackWriter:
          self._require_objcache()
          return self.objcache.exists(id, want_source=want_source)
  
          self._require_objcache()
          return self.objcache.exists(id, want_source=want_source)
  
+    def just_write(self, sha, type, content):
+        """Write an object to the pack file, bypassing the objcache.  Fails if
+        sha exists()."""
+        self._write(sha, type, content)
+
      def maybe_write(self, type, content):
          """Write an object to the pack file if not present and return its id."""
      def maybe_write(self, type, content):
          """Write an object to the pack file if not present and return its id."""
-        self._require_objcache()
          sha = calc_hash(type, content)
          if not self.exists(sha):
          sha = calc_hash(type, content)
          if not self.exists(sha):
-            self._write(sha, type, content)
+            self.just_write(sha, type, content)
+            self._require_objcache()
              self.objcache.add(sha)
          return sha
  
              self.objcache.add(sha)
          return sha
  
@@ -544,77 +732,93 @@ class PackWriter:
  
      def new_tree(self, shalist):
          """Create a tree object in the pack."""
  
      def new_tree(self, shalist):
          """Create a tree object in the pack."""
-        shalist = sorted(shalist, key = _shalist_sort_key)
-        l = []
-        for (mode,name,bin) in shalist:
-            assert(mode)
-            assert(mode != '0')
-            assert(mode[0] != '0')
-            assert(name)
-            assert(len(bin) == 20)
-            l.append('%s %s\0%s' % (mode,name,bin))
-        return self.maybe_write('tree', ''.join(l))
-
-    def _new_commit(self, tree, parent, author, adate, committer, cdate, msg):
+        content = tree_encode(shalist)
+        return self.maybe_write('tree', content)
+
+    def new_commit(self, tree, parent,
+                   author, adate_sec, adate_tz,
+                   committer, cdate_sec, cdate_tz,
+                   msg):
+        """Create a commit object in the pack.  The date_sec values must be
+        epoch-seconds, and if a tz is None, the local timezone is assumed."""
+        if adate_tz:
+            adate_str = _git_date_str(adate_sec, adate_tz)
+        else:
+            adate_str = _local_git_date_str(adate_sec)
+        if cdate_tz:
+            cdate_str = _git_date_str(cdate_sec, cdate_tz)
+        else:
+            cdate_str = _local_git_date_str(cdate_sec)
          l = []
          if tree: l.append('tree %s' % tree.encode('hex'))
          if parent: l.append('parent %s' % parent.encode('hex'))
          l = []
          if tree: l.append('tree %s' % tree.encode('hex'))
          if parent: l.append('parent %s' % parent.encode('hex'))
-        if author: l.append('author %s %s' % (author, _git_date(adate)))
-        if committer: l.append('committer %s %s' % (committer, _git_date(cdate)))
+        if author: l.append('author %s %s' % (author, adate_str))
+        if committer: l.append('committer %s %s' % (committer, cdate_str))
          l.append('')
          l.append(msg)
          return self.maybe_write('commit', '\n'.join(l))
  
          l.append('')
          l.append(msg)
          return self.maybe_write('commit', '\n'.join(l))
  
-    def new_commit(self, parent, tree, date, msg):
-        """Create a commit object in the pack."""
-        userline = '%s <%s@%s>' % (userfullname(), username(), hostname())
-        commit = self._new_commit(tree, parent,
-                                  userline, date, userline, date,
-                                  msg)
-        return commit
-
      def abort(self):
          """Remove the pack file from disk."""
          f = self.file
          if f:
      def abort(self):
          """Remove the pack file from disk."""
          f = self.file
          if f:
-            self.idx = None
+            pfd = self.parentfd
              self.file = None
              self.file = None
-            f.close()
-            os.unlink(self.filename + '.pack')
+            self.parentfd = None
+            self.idx = None
+            try:
+                try:
+                    os.unlink(self.filename + '.pack')
+                finally:
+                    f.close()
+            finally:
+                if pfd is not None:
+                    os.close(pfd)
  
      def _end(self, run_midx=True):
          f = self.file
          if not f: return None
          self.file = None
  
      def _end(self, run_midx=True):
          f = self.file
          if not f: return None
          self.file = None
-        self.objcache = None
-        idx = self.idx
-        self.idx = None
+        try:
+            self.objcache = None
+            idx = self.idx
+            self.idx = None
  
  
-        # update object count
-        f.seek(8)
-        cp = struct.pack('!i', self.count)
-        assert(len(cp) == 4)
-        f.write(cp)
-
-        # calculate the pack sha1sum
-        f.seek(0)
-        sum = Sha1()
-        for b in chunkyreader(f):
-            sum.update(b)
-        packbin = sum.digest()
-        f.write(packbin)
-        f.close()
+            # update object count
+            f.seek(8)
+            cp = struct.pack('!i', self.count)
+            assert(len(cp) == 4)
+            f.write(cp)
+
+            # calculate the pack sha1sum
+            f.seek(0)
+            sum = Sha1()
+            for b in chunkyreader(f):
+                sum.update(b)
+            packbin = sum.digest()
+            f.write(packbin)
+            fdatasync(f.fileno())
+        finally:
+            f.close()
  
          obj_list_sha = self._write_pack_idx_v2(self.filename + '.idx', idx, packbin)
  
          obj_list_sha = self._write_pack_idx_v2(self.filename + '.idx', idx, packbin)
-
-        nameprefix = repo('objects/pack/pack-%s' % obj_list_sha)
+        nameprefix = os.path.join(self.repo_dir,
+                                  'objects/pack/pack-' +  obj_list_sha)
          if os.path.exists(self.filename + '.map'):
              os.unlink(self.filename + '.map')
          os.rename(self.filename + '.pack', nameprefix + '.pack')
          os.rename(self.filename + '.idx', nameprefix + '.idx')
          if os.path.exists(self.filename + '.map'):
              os.unlink(self.filename + '.map')
          os.rename(self.filename + '.pack', nameprefix + '.pack')
          os.rename(self.filename + '.idx', nameprefix + '.idx')
+        try:
+            os.fsync(self.parentfd)
+        finally:
+            os.close(self.parentfd)
  
          if run_midx:
  
          if run_midx:
-            auto_midx(repo('objects/pack'))
+            auto_midx(os.path.join(self.repo_dir, 'objects/pack'))
+
+        if self.on_pack_finish:
+            self.on_pack_finish(nameprefix)
+
          return nameprefix
  
      def close(self, run_midx=True):
          return nameprefix
  
      def close(self, run_midx=True):
@@ -622,54 +826,80 @@ class PackWriter:
          return self._end(run_midx=run_midx)
  
      def _write_pack_idx_v2(self, filename, idx, packbin):
          return self._end(run_midx=run_midx)
  
      def _write_pack_idx_v2(self, filename, idx, packbin):
+        ofs64_count = 0
+        for section in idx:
+            for entry in section:
+                if entry[2] >= 2**31:
+                    ofs64_count += 1
+
+        # Length: header + fan-out + shas-and-crcs + overflow-offsets
+        index_len = 8 + (4 * 256) + (28 * self.count) + (8 * ofs64_count)
+        idx_map = None
          idx_f = open(filename, 'w+b')
          idx_f = open(filename, 'w+b')
-        idx_f.write('\377tOc\0\0\0\2')
-
-        ofs64_ofs = 8 + 4*256 + 28*self.count
-        idx_f.truncate(ofs64_ofs)
-        idx_f.seek(0)
-        idx_map = mmap_readwrite(idx_f, close=False)
-        idx_f.seek(0, SEEK_END)
-        count = _helpers.write_idx(idx_f, idx_map, idx, self.count)
-        assert(count == self.count)
-        idx_map.close()
-        idx_f.write(packbin)
-
-        idx_f.seek(0)
-        idx_sum = Sha1()
-        b = idx_f.read(8 + 4*256)
-        idx_sum.update(b)
-
-        obj_list_sum = Sha1()
-        for b in chunkyreader(idx_f, 20*self.count):
-            idx_sum.update(b)
-            obj_list_sum.update(b)
-        namebase = obj_list_sum.hexdigest()
-
-        for b in chunkyreader(idx_f):
+        try:
+            idx_f.truncate(index_len)
+            fdatasync(idx_f.fileno())
+            idx_map = mmap_readwrite(idx_f, close=False)
+            try:
+                count = _helpers.write_idx(filename, idx_map, idx, self.count)
+                assert(count == self.count)
+                idx_map.flush()
+            finally:
+                idx_map.close()
+        finally:
+            idx_f.close()
+
+        idx_f = open(filename, 'a+b')
+        try:
+            idx_f.write(packbin)
+            idx_f.seek(0)
+            idx_sum = Sha1()
+            b = idx_f.read(8 + 4*256)
              idx_sum.update(b)
              idx_sum.update(b)
-        idx_f.write(idx_sum.digest())
-        idx_f.close()
-
-        return namebase
-
-
-def _git_date(date):
-    return '%d %s' % (date, time.strftime('%z', time.localtime(date)))
-
  
  
-def _gitenv():
-    os.environ['GIT_DIR'] = os.path.abspath(repo())
+            obj_list_sum = Sha1()
+            for b in chunkyreader(idx_f, 20*self.count):
+                idx_sum.update(b)
+                obj_list_sum.update(b)
+            namebase = obj_list_sum.hexdigest()
+
+            for b in chunkyreader(idx_f):
+                idx_sum.update(b)
+            idx_f.write(idx_sum.digest())
+            fdatasync(idx_f.fileno())
+            return namebase
+        finally:
+            idx_f.close()
+
+
+def _gitenv(repo_dir = None):
+    if not repo_dir:
+        repo_dir = repo()
+    def env():
+        os.environ['GIT_DIR'] = os.path.abspath(repo_dir)
+    return env
+
+
+def list_refs(refnames=None, repo_dir=None,
+              limit_to_heads=False, limit_to_tags=False):
+    """Yield (refname, hash) tuples for all repository refs unless
+    refnames are specified.  In that case, only include tuples for
+    those refs.  The limits restrict the result items to refs/heads or
+    refs/tags.  If both limits are specified, items from both sources
+    will be included.
  
  
-
-def list_refs(refname = None):
-    """Generate a list of tuples in the form (refname,hash).
-    If a ref name is specified, list only this particular ref.
      """
      """
-    argv = ['git', 'show-ref', '--']
-    if refname:
-        argv += [refname]
-    p = subprocess.Popen(argv, preexec_fn = _gitenv, stdout = subprocess.PIPE)
+    argv = ['git', 'show-ref']
+    if limit_to_heads:
+        argv.append('--heads')
+    if limit_to_tags:
+        argv.append('--tags')
+    argv.append('--')
+    if refnames:
+        argv += refnames
+    p = subprocess.Popen(argv,
+                         preexec_fn = _gitenv(repo_dir),
+                         stdout = subprocess.PIPE)
      out = p.stdout.read().strip()
      rv = p.wait()  # not fatal
      if rv:
      out = p.stdout.read().strip()
      rv = p.wait()  # not fatal
      if rv:
@@ -680,9 +910,10 @@ def list_refs(refname = None):
              yield (name, sha.decode('hex'))
  
  
              yield (name, sha.decode('hex'))
  
  
-def read_ref(refname):
+def read_ref(refname, repo_dir = None):
      """Get the commit id of the most recent commit made on a given ref."""
      """Get the commit id of the most recent commit made on a given ref."""
-    l = list(list_refs(refname))
+    refs = list_refs(refnames=[refname], repo_dir=repo_dir, limit_to_heads=True)
+    l = tuple(islice(refs, 2))
      if l:
          assert(len(l) == 1)
          return l[0][1]
      if l:
          assert(len(l) == 1)
          return l[0][1]
@@ -690,7 +921,7 @@ def read_ref(refname):
          return None
  
  
          return None
  
  
-def rev_list(ref, count=None):
+def rev_list(ref, count=None, repo_dir=None):
      """Generate a list of reachable commits in reverse chronological order.
  
      This generator walks through commits, from child to parent, that are
      """Generate a list of reachable commits in reverse chronological order.
  
      This generator walks through commits, from child to parent, that are
@@ -704,8 +935,10 @@ def rev_list(ref, count=None):
      opts = []
      if count:
          opts += ['-n', str(atoi(count))]
      opts = []
      if count:
          opts += ['-n', str(atoi(count))]
-    argv = ['git', 'rev-list', '--pretty=format:%ct'] + opts + [ref, '--']
-    p = subprocess.Popen(argv, preexec_fn = _gitenv, stdout = subprocess.PIPE)
+    argv = ['git', 'rev-list', '--pretty=format:%at'] + opts + [ref, '--']
+    p = subprocess.Popen(argv,
+                         preexec_fn = _gitenv(repo_dir),
+                         stdout = subprocess.PIPE)
      commit = None
      for row in p.stdout:
          s = row.strip()
      commit = None
      for row in p.stdout:
          s = row.strip()
@@ -719,14 +952,18 @@ def rev_list(ref, count=None):
          raise GitError, 'git rev-list returned error %d' % rv
  
  
          raise GitError, 'git rev-list returned error %d' % rv
  
  
-def rev_get_date(ref):
-    """Get the date of the latest commit on the specified ref."""
-    for (date, commit) in rev_list(ref, count=1):
-        return date
-    raise GitError, 'no such commit %r' % ref
+def get_commit_dates(refs, repo_dir=None):
+    """Get the dates for the specified commit refs.  For now, every unique
+       string in refs must resolve to a different commit or this
+       function will fail."""
+    result = []
+    for ref in refs:
+        commit = get_commit_items(ref, cp(repo_dir))
+        result.append(commit.author_sec)
+    return result
  
  
  
  
-def rev_parse(committish):
+def rev_parse(committish, repo_dir=None):
      """Resolve the full hash for 'committish', if it exists.
  
      Should be roughly equivalent to 'git rev-parse'.
      """Resolve the full hash for 'committish', if it exists.
  
      Should be roughly equivalent to 'git rev-parse'.
@@ -734,12 +971,12 @@ def rev_parse(committish):
      Returns the hex value of the hash if it is found, None if 'committish' does
      not correspond to anything.
      """
      Returns the hex value of the hash if it is found, None if 'committish' does
      not correspond to anything.
      """
-    head = read_ref(committish)
+    head = read_ref(committish, repo_dir=repo_dir)
      if head:
          debug2("resolved from ref: commit = %s\n" % head.encode('hex'))
          return head
  
      if head:
          debug2("resolved from ref: commit = %s\n" % head.encode('hex'))
          return head
  
-    pL = PackIdxList(repo('objects/pack'))
+    pL = PackIdxList(repo('objects/pack', repo_dir=repo_dir))
  
      if len(committish) == 40:
          try:
  
      if len(committish) == 40:
          try:
@@ -753,14 +990,24 @@ def rev_parse(committish):
      return None
  
  
      return None
  
  
-def update_ref(refname, newval, oldval):
-    """Change the commit pointed to by a branch."""
+def update_ref(refname, newval, oldval, repo_dir=None):
+    """Update a repository reference."""
      if not oldval:
          oldval = ''
      if not oldval:
          oldval = ''
-    assert(refname.startswith('refs/heads/'))
+    assert(refname.startswith('refs/heads/') \
+           or refname.startswith('refs/tags/'))
      p = subprocess.Popen(['git', 'update-ref', refname,
                            newval.encode('hex'), oldval.encode('hex')],
      p = subprocess.Popen(['git', 'update-ref', refname,
                            newval.encode('hex'), oldval.encode('hex')],
-                         preexec_fn = _gitenv)
+                         preexec_fn = _gitenv(repo_dir))
+    _git_wait('git update-ref', p)
+
+
+def delete_ref(refname, oldvalue=None):
+    """Delete a repository reference (see git update-ref(1))."""
+    assert(refname.startswith('refs/'))
+    oldvalue = [] if not oldvalue else [oldvalue]
+    p = subprocess.Popen(['git', 'update-ref', '-d', refname] + oldvalue,
+                         preexec_fn = _gitenv())
      _git_wait('git update-ref', p)
  
  
      _git_wait('git update-ref', p)
  
  
@@ -788,42 +1035,36 @@ def init_repo(path=None):
      if parent and not os.path.exists(parent):
          raise GitError('parent directory "%s" does not exist\n' % parent)
      if os.path.exists(d) and not os.path.isdir(os.path.join(d, '.')):
      if parent and not os.path.exists(parent):
          raise GitError('parent directory "%s" does not exist\n' % parent)
      if os.path.exists(d) and not os.path.isdir(os.path.join(d, '.')):
-        raise GitError('"%d" exists but is not a directory\n' % d)
+        raise GitError('"%s" exists but is not a directory\n' % d)
      p = subprocess.Popen(['git', '--bare', 'init'], stdout=sys.stderr,
      p = subprocess.Popen(['git', '--bare', 'init'], stdout=sys.stderr,
-                         preexec_fn = _gitenv)
+                         preexec_fn = _gitenv())
      _git_wait('git init', p)
      # Force the index version configuration in order to ensure bup works
      # regardless of the version of the installed Git binary.
      p = subprocess.Popen(['git', 'config', 'pack.indexVersion', '2'],
      _git_wait('git init', p)
      # Force the index version configuration in order to ensure bup works
      # regardless of the version of the installed Git binary.
      p = subprocess.Popen(['git', 'config', 'pack.indexVersion', '2'],
-                         stdout=sys.stderr, preexec_fn = _gitenv)
+                         stdout=sys.stderr, preexec_fn = _gitenv())
+    _git_wait('git config', p)
+    # Enable the reflog
+    p = subprocess.Popen(['git', 'config', 'core.logAllRefUpdates', 'true'],
+                         stdout=sys.stderr, preexec_fn = _gitenv())
      _git_wait('git config', p)
  
  
  def check_repo_or_die(path=None):
      _git_wait('git config', p)
  
  
  def check_repo_or_die(path=None):
-    """Make sure a bup repository exists, and abort if not.
-    If the path to a particular repository was not specified, this function
-    initializes the default repository automatically.
-    """
+    """Check to see if a bup repository probably exists, and abort if not."""
      guess_repo(path)
      guess_repo(path)
-    if not os.path.isdir(repo('objects/pack/.')):
-        if repodir == home_repodir:
-            init_repo()
-        else:
-            log('error: %r is not a bup/git repository\n' % repo())
+    top = repo()
+    pst = stat_if_exists(top + '/objects/pack')
+    if pst and stat.S_ISDIR(pst.st_mode):
+        return
+    if not pst:
+        top_st = stat_if_exists(top)
+        if not top_st:
+            log('error: repository %r does not exist (see "bup help init")\n'
+                % top)
              sys.exit(15)
              sys.exit(15)
-
-
-def treeparse(buf):
-    """Generate a list of (mode, name, hash) tuples of objects from 'buf'."""
-    ofs = 0
-    while ofs < len(buf):
-        z = buf[ofs:].find('\0')
-        assert(z > 0)
-        spl = buf[ofs:ofs+z].split(' ', 1)
-        assert(len(spl) == 2)
-        sha = buf[ofs+z+1:ofs+z+1+20]
-        ofs += z+1+20
-        yield (spl[0], spl[1], sha)
+    log('error: %r is not a repository\n' % top)
+    sys.exit(14)
  
  
  _ver = None
  
  
  _ver = None
@@ -853,19 +1094,6 @@ def ver():
      return _ver
  
  
      return _ver
  
  
-def _git_wait(cmd, p):
-    rv = p.wait()
-    if rv != 0:
-        raise GitError('%s returned %d' % (cmd, rv))
-
-
-def _git_capture(argv):
-    p = subprocess.Popen(argv, stdout=subprocess.PIPE, preexec_fn = _gitenv)
-    r = p.stdout.read()
-    _git_wait(repr(argv), p)
-    return r
-
-
  class _AbortableIter:
      def __init__(self, it, onabort = None):
          self.it = it
  class _AbortableIter:
      def __init__(self, it, onabort = None):
          self.it = it
@@ -878,7 +1106,7 @@ class _AbortableIter:
      def next(self):
          try:
              return self.it.next()
      def next(self):
          try:
              return self.it.next()
-        except StopIteration, e:
+        except StopIteration as e:
              self.done = True
              raise
          except:
              self.done = True
              raise
          except:
@@ -896,11 +1124,18 @@ class _AbortableIter:
          self.abort()
  
  
          self.abort()
  
  
+class MissingObject(KeyError):
+    def __init__(self, id):
+        self.id = id
+        KeyError.__init__(self, 'object %r is missing' % id.encode('hex'))
+
+
  _ver_warned = 0
  class CatPipe:
      """Link to 'git cat-file' that is used to retrieve blob data."""
  _ver_warned = 0
  class CatPipe:
      """Link to 'git cat-file' that is used to retrieve blob data."""
-    def __init__(self):
+    def __init__(self, repo_dir = None):
          global _ver_warned
          global _ver_warned
+        self.repo_dir = repo_dir
          wanted = ('1','5','6')
          if ver() < wanted:
              if not _ver_warned:
          wanted = ('1','5','6')
          if ver() < wanted:
              if not _ver_warned:
@@ -919,22 +1154,23 @@ class CatPipe:
          self.p = None
          self.inprogress = None
  
          self.p = None
          self.inprogress = None
  
-    def _restart(self):
+    def restart(self):
          self._abort()
          self.p = subprocess.Popen(['git', 'cat-file', '--batch'],
                                    stdin=subprocess.PIPE,
                                    stdout=subprocess.PIPE,
                                    close_fds = True,
                                    bufsize = 4096,
          self._abort()
          self.p = subprocess.Popen(['git', 'cat-file', '--batch'],
                                    stdin=subprocess.PIPE,
                                    stdout=subprocess.PIPE,
                                    close_fds = True,
                                    bufsize = 4096,
-                                  preexec_fn = _gitenv)
+                                  preexec_fn = _gitenv(self.repo_dir))
  
      def _fast_get(self, id):
          if not self.p or self.p.poll() != None:
  
      def _fast_get(self, id):
          if not self.p or self.p.poll() != None:
-            self._restart()
+            self.restart()
          assert(self.p)
          assert(self.p)
-        assert(self.p.poll() == None)
+        poll_result = self.p.poll()
+        assert(poll_result == None)
          if self.inprogress:
          if self.inprogress:
-            log('_fast_get: opening %r while %r is open'
+            log('_fast_get: opening %r while %r is open\n'
                  % (id, self.inprogress))
          assert(not self.inprogress)
          assert(id.find('\n') < 0)
                  % (id, self.inprogress))
          assert(not self.inprogress)
          assert(id.find('\n') < 0)
@@ -946,7 +1182,7 @@ class CatPipe:
          hdr = self.p.stdout.readline()
          if hdr.endswith(' missing\n'):
              self.inprogress = None
          hdr = self.p.stdout.readline()
          if hdr.endswith(' missing\n'):
              self.inprogress = None
-            raise KeyError('blob %r is missing' % id)
+            raise MissingObject(id.decode('hex'))
          spl = hdr.split(' ')
          if len(spl) != 3 or len(spl[0]) != 40:
              raise GitError('expected blob, got %r' % spl)
          spl = hdr.split(' ')
          if len(spl) != 3 or len(spl[0]) != 40:
              raise GitError('expected blob, got %r' % spl)
@@ -958,9 +1194,10 @@ class CatPipe:
              yield type
              for blob in it:
                  yield blob
              yield type
              for blob in it:
                  yield blob
-            assert(self.p.stdout.readline() == '\n')
+            readline_result = self.p.stdout.readline()
+            assert(readline_result == '\n')
              self.inprogress = None
              self.inprogress = None
-        except Exception, e:
+        except Exception as e:
              it.abort()
              raise
  
              it.abort()
              raise
  
@@ -973,7 +1210,7 @@ class CatPipe:
  
          p = subprocess.Popen(['git', 'cat-file', type, id],
                               stdout=subprocess.PIPE,
  
          p = subprocess.Popen(['git', 'cat-file', type, id],
                               stdout=subprocess.PIPE,
-                             preexec_fn = _gitenv)
+                             preexec_fn = _gitenv(self.repo_dir))
          for blob in chunkyreader(p.stdout):
              yield blob
          _git_wait('git cat-file', p)
          for blob in chunkyreader(p.stdout):
              yield blob
          _git_wait('git cat-file', p)
@@ -985,7 +1222,7 @@ class CatPipe:
                  yield blob
          elif type == 'tree':
              treefile = ''.join(it)
                  yield blob
          elif type == 'tree':
              treefile = ''.join(it)
-            for (mode, name, sha) in treeparse(treefile):
+            for (mode, name, sha) in tree_decode(treefile):
                  for blob in self.join(sha.encode('hex')):
                      yield blob
          elif type == 'commit':
                  for blob in self.join(sha.encode('hex')):
                      yield blob
          elif type == 'commit':
@@ -1009,15 +1246,110 @@ class CatPipe:
          except StopIteration:
              log('booger!\n')
  
          except StopIteration:
              log('booger!\n')
  
-def tags():
+
+_cp = {}
+
+def cp(repo_dir=None):
+    """Create a CatPipe object or reuse the already existing one."""
+    global _cp, repodir
+    if not repo_dir:
+        repo_dir = repodir or repo()
+    repo_dir = os.path.abspath(repo_dir)
+    cp = _cp.get(repo_dir)
+    if not cp:
+        cp = CatPipe(repo_dir)
+        _cp[repo_dir] = cp
+    return cp
+
+
+def tags(repo_dir = None):
      """Return a dictionary of all tags in the form {hash: [tag_names, ...]}."""
      tags = {}
      """Return a dictionary of all tags in the form {hash: [tag_names, ...]}."""
      tags = {}
-    for (n,c) in list_refs():
-        if n.startswith('refs/tags/'):
-            name = n[10:]
-            if not c in tags:
-                tags[c] = []
+    for n, c in list_refs(repo_dir = repo_dir, limit_to_tags=True):
+        assert(n.startswith('refs/tags/'))
+        name = n[10:]
+        if not c in tags:
+            tags[c] = []
+        tags[c].append(name)  # more than one tag can point at 'c'
+    return tags
  
  
-            tags[c].append(name)  # more than one tag can point at 'c'
  
  
-    return tags
+WalkItem = namedtuple('WalkItem', ['id', 'type', 'mode',
+                                   'path', 'chunk_path', 'data'])
+# The path is the mangled path, and if an item represents a fragment
+# of a chunked file, the chunk_path will be the chunked subtree path
+# for the chunk, i.e. ['', '2d3115e', ...].  The top-level path for a
+# chunked file will have a chunk_path of [''].  So some chunk subtree
+# of the file '/foo/bar/baz' might look like this:
+#
+#   item.path = ['foo', 'bar', 'baz.bup']
+#   item.chunk_path = ['', '2d3115e', '016b097']
+#   item.type = 'tree'
+#   ...
+
+
+def walk_object(cat_pipe, id,
+                stop_at=None,
+                include_data=None):
+    """Yield everything reachable from id via cat_pipe as a WalkItem,
+    stopping whenever stop_at(id) returns true.  Throw MissingObject
+    if a hash encountered is missing from the repository, and don't
+    read or return blob content in the data field unless include_data
+    is set.
+    """
+    # Maintain the pending stack on the heap to avoid stack overflow
+    pending = [(id, [], [], None)]
+    while len(pending):
+        id, parent_path, chunk_path, mode = pending.pop()
+        if stop_at and stop_at(id):
+            continue
+
+        if (not include_data) and mode and stat.S_ISREG(mode):
+            # If the object is a "regular file", then it's a leaf in
+            # the graph, so we can skip reading the data if the caller
+            # hasn't requested it.
+            yield WalkItem(id=id, type='blob',
+                           chunk_path=chunk_path, path=parent_path,
+                           mode=mode,
+                           data=None)
+            continue
+
+        item_it = cat_pipe.get(id)
+        type = item_it.next()
+        if type not in ('blob', 'commit', 'tree'):
+            raise Exception('unexpected repository object type %r' % type)
+
+        # FIXME: set the mode based on the type when the mode is None
+        if type == 'blob' and not include_data:
+            # Dump data until we can ask cat_pipe not to fetch it
+            for ignored in item_it:
+                pass
+            data = None
+        else:
+            data = ''.join(item_it)
+
+        yield WalkItem(id=id, type=type,
+                       chunk_path=chunk_path, path=parent_path,
+                       mode=mode,
+                       data=(data if include_data else None))
+
+        if type == 'commit':
+            commit_items = parse_commit(data)
+            for pid in commit_items.parents:
+                pending.append((pid, parent_path, chunk_path, mode))
+            pending.append((commit_items.tree, parent_path, chunk_path,
+                            hashsplit.GIT_MODE_TREE))
+        elif type == 'tree':
+            for mode, name, ent_id in tree_decode(data):
+                demangled, bup_type = demangle_name(name, mode)
+                if chunk_path:
+                    sub_path = parent_path
+                    sub_chunk_path = chunk_path + [name]
+                else:
+                    sub_path = parent_path + [name]
+                    if bup_type == BUP_CHUNKED:
+                        sub_chunk_path = ['']
+                    else:
+                        sub_chunk_path = chunk_path
+                pending.append((ent_id.encode('hex'), sub_path, sub_chunk_path,
+                                mode))