upstream/mercurial-mirror Commit - r51397:862e3a13

store: rename `datafiles` to `data_entries`...

marmoute -

r51397:862e3a13 default

parent child

hgext/largefiles/lfutil.py

0 +1 -1

              # Copyright 2009-2010 Gregory P. Ward
              # Copyright 2009-2010 Intelerad Medical Systems Incorporated
              # Copyright 2010-2011 Fog Creek Software
              # Copyright 2010-2011 Unity Technologies
              #
              # This software may be used and distributed according to the terms of the
              # GNU General Public License version 2 or any later version.
              '''largefiles utility code: must not import other modules in this package.'''
              import contextlib
              import copy
              import os
              import stat
              from mercurial.i18n import _
              from mercurial.node import hex
              from mercurial.pycompat import open
              from mercurial import (
                  dirstate,
                  encoding,
                  error,
                  httpconnection,
                  match as matchmod,
                  pycompat,
                  requirements,
                  scmutil,
                  sparse,
                  util,
                  vfs as vfsmod,
              )
              from mercurial.utils import hashutil
              from mercurial.dirstateutils import timestamp
              shortname = b'.hglf'
              shortnameslash = shortname + b'/'
              longname = b'largefiles'
              # -- Private worker functions ------------------------------------------
              @contextlib.contextmanager
              def lfstatus(repo, value=True):
                  oldvalue = getattr(repo, 'lfstatus', False)
                  repo.lfstatus = value
                  try:
                      yield
                  finally:
                      repo.lfstatus = oldvalue
              def getminsize(ui, assumelfiles, opt, default=10):
                  lfsize = opt
                  if not lfsize and assumelfiles:
                      lfsize = ui.config(longname, b'minsize', default=default)
                  if lfsize:
                      try:
                          lfsize = float(lfsize)
                      except ValueError:
                          raise error.Abort(
                              _(b'largefiles: size must be number (not %s)\n') % lfsize
                          )
                  if lfsize is None:
                      raise error.Abort(_(b'minimum size for largefiles must be specified'))
                  return lfsize
              def link(src, dest):
                  """Try to create hardlink - if that fails, efficiently make a copy."""
                  util.makedirs(os.path.dirname(dest))
                  try:
                      util.oslink(src, dest)
                  except OSError:
                      # if hardlinks fail, fallback on atomic copy
                      with open(src, b'rb') as srcf, util.atomictempfile(dest) as dstf:
                          for chunk in util.filechunkiter(srcf):
                              dstf.write(chunk)
                      os.chmod(dest, os.stat(src).st_mode)
              def usercachepath(ui, hash):
                  """Return the correct location in the "global" largefiles cache for a file
                  with the given hash.
                  This cache is used for sharing of largefiles across repositories - both
                  to preserve download bandwidth and storage space."""
                  return os.path.join(_usercachedir(ui), hash)
              def _usercachedir(ui, name=longname):
                  '''Return the location of the "global" largefiles cache.'''
                  path = ui.configpath(name, b'usercache')
                  if path:
                      return path
                  hint = None
                  if pycompat.iswindows:
                      appdata = encoding.environ.get(
                          b'LOCALAPPDATA', encoding.environ.get(b'APPDATA')
                      )
                      if appdata:
                          return os.path.join(appdata, name)
                      hint = _(b"define %s or %s in the environment, or set %s.usercache") % (
                          b"LOCALAPPDATA",
                          b"APPDATA",
                          name,
                      )
                  elif pycompat.isdarwin:
                      home = encoding.environ.get(b'HOME')
                      if home:
                          return os.path.join(home, b'Library', b'Caches', name)
                      hint = _(b"define %s in the environment, or set %s.usercache") % (
                          b"HOME",
                          name,
                      )
                  elif pycompat.isposix:
                      path = encoding.environ.get(b'XDG_CACHE_HOME')
                      if path:
                          return os.path.join(path, name)
                      home = encoding.environ.get(b'HOME')
                      if home:
                          return os.path.join(home, b'.cache', name)
                      hint = _(b"define %s or %s in the environment, or set %s.usercache") % (
                          b"XDG_CACHE_HOME",
                          b"HOME",
                          name,
                      )
                  else:
                      raise error.Abort(
                          _(b'unknown operating system: %s\n') % pycompat.osname
                      )
                  raise error.Abort(_(b'unknown %s usercache location') % name, hint=hint)
              def inusercache(ui, hash):
                  path = usercachepath(ui, hash)
                  return os.path.exists(path)
              def findfile(repo, hash):
                  """Return store path of the largefile with the specified hash.
                  As a side effect, the file might be linked from user cache.
                  Return None if the file can't be found locally."""
                  path, exists = findstorepath(repo, hash)
                  if exists:
                      repo.ui.note(_(b'found %s in store\n') % hash)
                      return path
                  elif inusercache(repo.ui, hash):
                      repo.ui.note(_(b'found %s in system cache\n') % hash)
                      path = storepath(repo, hash)
                      link(usercachepath(repo.ui, hash), path)
                      return path
                  return None
              class largefilesdirstate(dirstate.dirstate):
                  _large_file_dirstate = True
                  _tr_key_suffix = b'-large-files'
                  def __getitem__(self, key):
                      return super(largefilesdirstate, self).__getitem__(unixpath(key))
                  def set_tracked(self, f):
                      return super(largefilesdirstate, self).set_tracked(unixpath(f))
                  def set_untracked(self, f):
                      return super(largefilesdirstate, self).set_untracked(unixpath(f))
                  def normal(self, f, parentfiledata=None):
                      # not sure if we should pass the `parentfiledata` down or throw it
                      # away. So throwing it away to stay on the safe side.
                      return super(largefilesdirstate, self).normal(unixpath(f))
                  def remove(self, f):
                      return super(largefilesdirstate, self).remove(unixpath(f))
                  def add(self, f):
                      return super(largefilesdirstate, self).add(unixpath(f))
                  def drop(self, f):
                      return super(largefilesdirstate, self).drop(unixpath(f))
                  def forget(self, f):
                      return super(largefilesdirstate, self).forget(unixpath(f))
                  def normallookup(self, f):
                      return super(largefilesdirstate, self).normallookup(unixpath(f))
                  def _ignore(self, f):
                      return False
                  def write(self, tr):
                      # (1) disable PENDING mode always
                      #     (lfdirstate isn't yet managed as a part of the transaction)
                      # (2) avoid develwarn 'use dirstate.write with ....'
                      if tr:
                          tr.addbackup(b'largefiles/dirstate', location=b'plain')
                      super(largefilesdirstate, self).write(None)
              def openlfdirstate(ui, repo, create=True):
                  """
                  Return a dirstate object that tracks largefiles: i.e. its root is
                  the repo root, but it is saved in .hg/largefiles/dirstate.
                  If a dirstate object already exists and is being used for a 'changing_*'
                  context, it will be returned.
                  """
                  sub_dirstate = getattr(repo.dirstate, '_sub_dirstate', None)
                  if sub_dirstate is not None:
                      return sub_dirstate
                  vfs = repo.vfs
                  lfstoredir = longname
                  opener = vfsmod.vfs(vfs.join(lfstoredir))
                  use_dirstate_v2 = requirements.DIRSTATE_V2_REQUIREMENT in repo.requirements
                  lfdirstate = largefilesdirstate(
                      opener,
                      ui,
                      repo.root,
                      repo.dirstate._validate,
                      lambda: sparse.matcher(repo),
                      repo.nodeconstants,
                      use_dirstate_v2,
                  )
                  # If the largefiles dirstate does not exist, populate and create
                  # it. This ensures that we create it on the first meaningful
                  # largefiles operation in a new clone.
                  if create and not vfs.exists(vfs.join(lfstoredir, b'dirstate')):
                      try:
                          with repo.wlock(wait=False), lfdirstate.changing_files(repo):
                              matcher = getstandinmatcher(repo)
                              standins = repo.dirstate.walk(
                                  matcher, subrepos=[], unknown=False, ignored=False
                              )
                              if len(standins) > 0:
                                  vfs.makedirs(lfstoredir)
                              for standin in standins:
                                  lfile = splitstandin(standin)
                                  lfdirstate.hacky_extension_update_file(
                                      lfile,
                                      p1_tracked=True,
                                      wc_tracked=True,
                                      possibly_dirty=True,
                                  )
                      except error.LockError:
                          # Assume that whatever was holding the lock was important.
                          # If we were doing something important, we would already have
                          # either the lock or a largefile dirstate.
                          pass
                  return lfdirstate
              def lfdirstatestatus(lfdirstate, repo):
                  pctx = repo[b'.']
                  match = matchmod.always()
                  unsure, s, mtime_boundary = lfdirstate.status(
                      match, subrepos=[], ignored=False, clean=False, unknown=False
                  )
                  modified, clean = s.modified, s.clean
                  wctx = repo[None]
                  for lfile in unsure:
                      try:
                          fctx = pctx[standin(lfile)]
                      except LookupError:
                          fctx = None
                      if not fctx or readasstandin(fctx) != hashfile(repo.wjoin(lfile)):
                          modified.append(lfile)
                      else:
                          clean.append(lfile)
                          st = wctx[lfile].lstat()
                          mode = st.st_mode
                          size = st.st_size
                          mtime = timestamp.reliable_mtime_of(st, mtime_boundary)
                          if mtime is not None:
                              cache_data = (mode, size, mtime)
                              lfdirstate.set_clean(lfile, cache_data)
                  return s
              def listlfiles(repo, rev=None, matcher=None):
                  """return a list of largefiles in the working copy or the
                  specified changeset"""
                  if matcher is None:
                      matcher = getstandinmatcher(repo)
                  # ignore unknown files in working directory
                  return [
                      splitstandin(f)
                      for f in repo[rev].walk(matcher)
                      if rev is not None or repo.dirstate.get_entry(f).any_tracked
                  ]
              def instore(repo, hash, forcelocal=False):
                  '''Return true if a largefile with the given hash exists in the store'''
                  return os.path.exists(storepath(repo, hash, forcelocal))
              def storepath(repo, hash, forcelocal=False):
                  """Return the correct location in the repository largefiles store for a
                  file with the given hash."""
                  if not forcelocal and repo.shared():
                      return repo.vfs.reljoin(repo.sharedpath, longname, hash)
                  return repo.vfs.join(longname, hash)
              def findstorepath(repo, hash):
                  """Search through the local store path(s) to find the file for the given
                  hash.  If the file is not found, its path in the primary store is returned.
                  The return value is a tuple of (path, exists(path)).
                  """
                  # For shared repos, the primary store is in the share source.  But for
                  # backward compatibility, force a lookup in the local store if it wasn't
                  # found in the share source.
                  path = storepath(repo, hash, False)
                  if instore(repo, hash):
                      return (path, True)
                  elif repo.shared() and instore(repo, hash, True):
                      return storepath(repo, hash, True), True
                  return (path, False)
              def copyfromcache(repo, hash, filename):
                  """Copy the specified largefile from the repo or system cache to
                  filename in the repository. Return true on success or false if the
                  file was not found in either cache (which should not happened:
                  this is meant to be called only after ensuring that the needed
                  largefile exists in the cache)."""
                  wvfs = repo.wvfs
                  path = findfile(repo, hash)
                  if path is None:
                      return False
                  wvfs.makedirs(wvfs.dirname(wvfs.join(filename)))
                  # The write may fail before the file is fully written, but we
                  # don't use atomic writes in the working copy.
                  with open(path, b'rb') as srcfd, wvfs(filename, b'wb') as destfd:
                      gothash = copyandhash(util.filechunkiter(srcfd), destfd)
                  if gothash != hash:
                      repo.ui.warn(
                          _(b'%s: data corruption in %s with hash %s\n')
                          % (filename, path, gothash)
                      )
                      wvfs.unlink(filename)
                      return False
                  return True
              def copytostore(repo, ctx, file, fstandin):
                  wvfs = repo.wvfs
                  hash = readasstandin(ctx[fstandin])
                  if instore(repo, hash):
                      return
                  if wvfs.exists(file):
                      copytostoreabsolute(repo, wvfs.join(file), hash)
                  else:
                      repo.ui.warn(
                          _(b"%s: largefile %s not available from local store\n")
                          % (file, hash)
                      )
              def copyalltostore(repo, node):
                  '''Copy all largefiles in a given revision to the store'''
                  ctx = repo[node]
                  for filename in ctx.files():
                      realfile = splitstandin(filename)
                      if realfile is not None and filename in ctx.manifest():
                          copytostore(repo, ctx, realfile, filename)
              def copytostoreabsolute(repo, file, hash):
                  if inusercache(repo.ui, hash):
                      link(usercachepath(repo.ui, hash), storepath(repo, hash))
                  else:
                      util.makedirs(os.path.dirname(storepath(repo, hash)))
                      with open(file, b'rb') as srcf:
                          with util.atomictempfile(
                              storepath(repo, hash), createmode=repo.store.createmode
                          ) as dstf:
                              for chunk in util.filechunkiter(srcf):
                                  dstf.write(chunk)
                      linktousercache(repo, hash)
              def linktousercache(repo, hash):
                  """Link / copy the largefile with the specified hash from the store
                  to the cache."""
                  path = usercachepath(repo.ui, hash)
                  link(storepath(repo, hash), path)
              def getstandinmatcher(repo, rmatcher=None):
                  '''Return a match object that applies rmatcher to the standin directory'''
                  wvfs = repo.wvfs
                  standindir = shortname
                  # no warnings about missing files or directories
                  badfn = lambda f, msg: None
                  if rmatcher and not rmatcher.always():
                      pats = [wvfs.join(standindir, pat) for pat in rmatcher.files()]
                      if not pats:
                          pats = [wvfs.join(standindir)]
                      match = scmutil.match(repo[None], pats, badfn=badfn)
                  else:
                      # no patterns: relative to repo root
                      match = scmutil.match(repo[None], [wvfs.join(standindir)], badfn=badfn)
                  return match
              def composestandinmatcher(repo, rmatcher):
                  """Return a matcher that accepts standins corresponding to the
                  files accepted by rmatcher. Pass the list of files in the matcher
                  as the paths specified by the user."""
                  smatcher = getstandinmatcher(repo, rmatcher)
                  isstandin = smatcher.matchfn
                  def composedmatchfn(f):
                      return isstandin(f) and rmatcher.matchfn(splitstandin(f))
                  smatcher.matchfn = composedmatchfn
                  return smatcher
              def standin(filename):
                  """Return the repo-relative path to the standin for the specified big
                  file."""
                  # Notes:
                  # 1) Some callers want an absolute path, but for instance addlargefiles
                  #    needs it repo-relative so it can be passed to repo[None].add().  So
                  #    leave it up to the caller to use repo.wjoin() to get an absolute path.
                  # 2) Join with '/' because that's what dirstate always uses, even on
                  #    Windows. Change existing separator to '/' first in case we are
                  #    passed filenames from an external source (like the command line).
                  return shortnameslash + util.pconvert(filename)
              def isstandin(filename):
                  """Return true if filename is a big file standin. filename must be
                  in Mercurial's internal form (slash-separated)."""
                  return filename.startswith(shortnameslash)
              def splitstandin(filename):
                  # Split on / because that's what dirstate always uses, even on Windows.
                  # Change local separator to / first just in case we are passed filenames
                  # from an external source (like the command line).
                  bits = util.pconvert(filename).split(b'/', 1)
                  if len(bits) == 2 and bits[0] == shortname:
                      return bits[1]
                  else:
                      return None
              def updatestandin(repo, lfile, standin):
                  """Re-calculate hash value of lfile and write it into standin
                  This assumes that "lfutil.standin(lfile) == standin", for efficiency.
                  """
                  file = repo.wjoin(lfile)
                  if repo.wvfs.exists(lfile):
                      hash = hashfile(file)
                      executable = getexecutable(file)
                      writestandin(repo, standin, hash, executable)
                  else:
                      raise error.Abort(_(b'%s: file not found!') % lfile)
              def readasstandin(fctx):
                  """read hex hash from given filectx of standin file
                  This encapsulates how "standin" data is stored into storage layer."""
                  return fctx.data().strip()
              def writestandin(repo, standin, hash, executable):
                  '''write hash to <repo.root>/<standin>'''
                  repo.wwrite(standin, hash + b'\n', executable and b'x' or b'')
              def copyandhash(instream, outfile):
                  """Read bytes from instream (iterable) and write them to outfile,
                  computing the SHA-1 hash of the data along the way. Return the hash."""
                  hasher = hashutil.sha1(b'')
                  for data in instream:
                      hasher.update(data)
                      outfile.write(data)
                  return hex(hasher.digest())
              def hashfile(file):
                  if not os.path.exists(file):
                      return b''
                  with open(file, b'rb') as fd:
                      return hexsha1(fd)
              def getexecutable(filename):
                  mode = os.stat(filename).st_mode
                  return (
                      (mode & stat.S_IXUSR)
                      and (mode & stat.S_IXGRP)
                      and (mode & stat.S_IXOTH)
                  )
              def urljoin(first, second, *arg):
                  def join(left, right):
                      if not left.endswith(b'/'):
                          left += b'/'
                      if right.startswith(b'/'):
                          right = right[1:]
                      return left + right
                  url = join(first, second)
                  for a in arg:
                      url = join(url, a)
                  return url
              def hexsha1(fileobj):
                  """hexsha1 returns the hex-encoded sha1 sum of the data in the file-like
                  object data"""
                  h = hashutil.sha1()
                  for chunk in util.filechunkiter(fileobj):
                      h.update(chunk)
                  return hex(h.digest())
              def httpsendfile(ui, filename):
                  return httpconnection.httpsendfile(ui, filename, b'rb')
              def unixpath(path):
                  '''Return a version of path normalized for use with the lfdirstate.'''
                  return util.pconvert(os.path.normpath(path))
              def islfilesrepo(repo):
                  '''Return true if the repo is a largefile repo.'''
                  if b'largefiles' in repo.requirements:
-                     for entry in repo.store.datafiles():
+                     for entry in repo.store.data_entries():
                          if entry.is_revlog and shortnameslash in entry.target_id:
                              return True
                  return any(openlfdirstate(repo.ui, repo, False))
              class storeprotonotcapable(Exception):
                  def __init__(self, storetypes):
                      self.storetypes = storetypes
              def getstandinsstate(repo):
                  standins = []
                  matcher = getstandinmatcher(repo)
                  wctx = repo[None]
                  for standin in repo.dirstate.walk(
                      matcher, subrepos=[], unknown=False, ignored=False
                  ):
                      lfile = splitstandin(standin)
                      try:
                          hash = readasstandin(wctx[standin])
                      except IOError:
                          hash = None
                      standins.append((lfile, hash))
                  return standins
              def synclfdirstate(repo, lfdirstate, lfile, normallookup):
                  lfstandin = standin(lfile)
                  if lfstandin not in repo.dirstate:
                      lfdirstate.hacky_extension_update_file(
                          lfile,
                          p1_tracked=False,
                          wc_tracked=False,
                      )
                  else:
                      entry = repo.dirstate.get_entry(lfstandin)
                      lfdirstate.hacky_extension_update_file(
                          lfile,
                          wc_tracked=entry.tracked,
                          p1_tracked=entry.p1_tracked,
                          p2_info=entry.p2_info,
                          possibly_dirty=True,
                      )
              def markcommitted(orig, ctx, node):
                  repo = ctx.repo()
                  with repo.dirstate.changing_parents(repo):
                      orig(node)
                      # ATTENTION: "ctx.files()" may differ from "repo[node].files()"
                      # because files coming from the 2nd parent are omitted in the latter.
                      #
                      # The former should be used to get targets of "synclfdirstate",
                      # because such files:
                      # - are marked as "a" by "patch.patch()" (e.g. via transplant), and
                      # - have to be marked as "n" after commit, but
                      # - aren't listed in "repo[node].files()"
                      lfdirstate = openlfdirstate(repo.ui, repo)
                      for f in ctx.files():
                          lfile = splitstandin(f)
                          if lfile is not None:
                              synclfdirstate(repo, lfdirstate, lfile, False)
                  # As part of committing, copy all of the largefiles into the cache.
                  #
                  # Using "node" instead of "ctx" implies additional "repo[node]"
                  # lookup while copyalltostore(), but can omit redundant check for
                  # files comming from the 2nd parent, which should exist in store
                  # at merging.
                  copyalltostore(repo, node)
              def getlfilestoupdate(oldstandins, newstandins):
                  changedstandins = set(oldstandins).symmetric_difference(set(newstandins))
                  filelist = []
                  for f in changedstandins:
                      if f[0] not in filelist:
                          filelist.append(f[0])
                  return filelist
              def getlfilestoupload(repo, missing, addfunc):
                  makeprogress = repo.ui.makeprogress
                  with makeprogress(
                      _(b'finding outgoing largefiles'),
                      unit=_(b'revisions'),
                      total=len(missing),
                  ) as progress:
                      for i, n in enumerate(missing):
                          progress.update(i)
                          parents = [p for p in repo[n].parents() if p != repo.nullid]
                          with lfstatus(repo, value=False):
                              ctx = repo[n]
                          files = set(ctx.files())
                          if len(parents) == 2:
                              mc = ctx.manifest()
                              mp1 = ctx.p1().manifest()
                              mp2 = ctx.p2().manifest()
                              for f in mp1:
                                  if f not in mc:
                                      files.add(f)
                              for f in mp2:
                                  if f not in mc:
                                      files.add(f)
                              for f in mc:
                                  if mc[f] != mp1.get(f, None) or mc[f] != mp2.get(f, None):
                                      files.add(f)
                          for fn in files:
                              if isstandin(fn) and fn in ctx:
                                  addfunc(fn, readasstandin(ctx[fn]))
              def updatestandinsbymatch(repo, match):
                  """Update standins in the working directory according to specified match
                  This returns (possibly modified) ``match`` object to be used for
                  subsequent commit process.
                  """
                  ui = repo.ui
                  # Case 1: user calls commit with no specific files or
                  # include/exclude patterns: refresh and commit all files that
                  # are "dirty".
                  if match is None or match.always():
                      # Spend a bit of time here to get a list of files we know
                      # are modified so we can compare only against those.
                      # It can cost a lot of time (several seconds)
                      # otherwise to update all standins if the largefiles are
                      # large.
                      dirtymatch = matchmod.always()
                      with repo.dirstate.running_status(repo):
                          lfdirstate = openlfdirstate(ui, repo)
                          unsure, s, mtime_boundary = lfdirstate.status(
                              dirtymatch,
                              subrepos=[],
                              ignored=False,
                              clean=False,
                              unknown=False,
                          )
                      modifiedfiles = unsure + s.modified + s.added + s.removed
                      lfiles = listlfiles(repo)
                      # this only loops through largefiles that exist (not
                      # removed/renamed)
                      for lfile in lfiles:
                          if lfile in modifiedfiles:
                              fstandin = standin(lfile)
                              if repo.wvfs.exists(fstandin):
                                  # this handles the case where a rebase is being
                                  # performed and the working copy is not updated
                                  # yet.
                                  if repo.wvfs.exists(lfile):
                                      updatestandin(repo, lfile, fstandin)
                      return match
                  lfiles = listlfiles(repo)
                  match._files = repo._subdirlfs(match.files(), lfiles)
                  # Case 2: user calls commit with specified patterns: refresh
                  # any matching big files.
                  smatcher = composestandinmatcher(repo, match)
                  standins = repo.dirstate.walk(
                      smatcher, subrepos=[], unknown=False, ignored=False
                  )
                  # No matching big files: get out of the way and pass control to
                  # the usual commit() method.
                  if not standins:
                      return match
                  # Refresh all matching big files.  It's possible that the
                  # commit will end up failing, in which case the big files will
                  # stay refreshed.  No harm done: the user modified them and
                  # asked to commit them, so sooner or later we're going to
                  # refresh the standins.  Might as well leave them refreshed.
                  lfdirstate = openlfdirstate(ui, repo)
                  for fstandin in standins:
                      lfile = splitstandin(fstandin)
                      if lfdirstate.get_entry(lfile).tracked:
                          updatestandin(repo, lfile, fstandin)
                  # Cook up a new matcher that only matches regular files or
                  # standins corresponding to the big files requested by the
                  # user.  Have to modify _files to prevent commit() from
                  # complaining "not tracked" for big files.
                  match = copy.copy(match)
                  origmatchfn = match.matchfn
                  # Check both the list of largefiles and the list of
                  # standins because if a largefile was removed, it
                  # won't be in the list of largefiles at this point
                  match._files += sorted(standins)
                  actualfiles = []
                  for f in match._files:
                      fstandin = standin(f)
                      # For largefiles, only one of the normal and standin should be
                      # committed (except if one of them is a remove).  In the case of a
                      # standin removal, drop the normal file if it is unknown to dirstate.
                      # Thus, skip plain largefile names but keep the standin.
                      if f in lfiles or fstandin in standins:
                          if not repo.dirstate.get_entry(fstandin).removed:
                              if not repo.dirstate.get_entry(f).removed:
                                  continue
                          elif not repo.dirstate.get_entry(f).any_tracked:
                              continue
                      actualfiles.append(f)
                  match._files = actualfiles
                  def matchfn(f):
                      if origmatchfn(f):
                          return f not in lfiles
                      else:
                          return f in standins
                  match.matchfn = matchfn
                  return match
              class automatedcommithook:
                  """Stateful hook to update standins at the 1st commit of resuming
                  For efficiency, updating standins in the working directory should
                  be avoided while automated committing (like rebase, transplant and
                  so on), because they should be updated before committing.
                  But the 1st commit of resuming automated committing (e.g. ``rebase
                  --continue``) should update them, because largefiles may be
                  modified manually.
                  """
                  def __init__(self, resuming):
                      self.resuming = resuming
                  def __call__(self, repo, match):
                      if self.resuming:
                          self.resuming = False  # avoids updating at subsequent commits
                          return updatestandinsbymatch(repo, match)
                      else:
                          return match
              def getstatuswriter(ui, repo, forcibly=None):
                  """Return the function to write largefiles specific status out
                  If ``forcibly`` is ``None``, this returns the last element of
                  ``repo._lfstatuswriters`` as "default" writer function.
                  Otherwise, this returns the function to always write out (or
                  ignore if ``not forcibly``) status.
                  """
                  if forcibly is None and util.safehasattr(repo, b'_largefilesenabled'):
                      return repo._lfstatuswriters[-1]
                  else:
                      if forcibly:
                          return ui.status  # forcibly WRITE OUT
                      else:
                          return lambda *msg, **opts: None  # forcibly IGNORE

hgext/largefiles/reposetup.py

0 +1 -1

              # Copyright 2009-2010 Gregory P. Ward
              # Copyright 2009-2010 Intelerad Medical Systems Incorporated
              # Copyright 2010-2011 Fog Creek Software
              # Copyright 2010-2011 Unity Technologies
              #
              # This software may be used and distributed according to the terms of the
              # GNU General Public License version 2 or any later version.
              '''setup for largefiles repositories: reposetup'''
              import copy
              from mercurial.i18n import _
              from mercurial import (
                  error,
                  extensions,
                  localrepo,
                  match as matchmod,
                  scmutil,
                  util,
              )
              from mercurial.dirstateutils import timestamp
              from . import (
                  lfcommands,
                  lfutil,
              )
              def reposetup(ui, repo):
                  # wire repositories should be given new wireproto functions
                  # by "proto.wirereposetup()" via "hg.wirepeersetupfuncs"
                  if not repo.local():
                      return
                  class lfilesrepo(repo.__class__):
                      # the mark to examine whether "repo" object enables largefiles or not
                      _largefilesenabled = True
                      lfstatus = False
                      # When lfstatus is set, return a context that gives the names
                      # of largefiles instead of their corresponding standins and
                      # identifies the largefiles as always binary, regardless of
                      # their actual contents.
                      def __getitem__(self, changeid):
                          ctx = super(lfilesrepo, self).__getitem__(changeid)
                          if self.lfstatus:
                              def files(orig):
                                  filenames = orig()
                                  return [lfutil.splitstandin(f) or f for f in filenames]
                              extensions.wrapfunction(ctx, 'files', files)
                              def manifest(orig):
                                  man1 = orig()
                                  class lfilesmanifest(man1.__class__):
                                      def __contains__(self, filename):
                                          orig = super(lfilesmanifest, self).__contains__
                                          return orig(filename) or orig(
                                              lfutil.standin(filename)
                                          )
                                  man1.__class__ = lfilesmanifest
                                  return man1
                              extensions.wrapfunction(ctx, 'manifest', manifest)
                              def filectx(orig, path, fileid=None, filelog=None):
                                  try:
                                      if filelog is not None:
                                          result = orig(path, fileid, filelog)
                                      else:
                                          result = orig(path, fileid)
                                  except error.LookupError:
                                      # Adding a null character will cause Mercurial to
                                      # identify this as a binary file.
                                      if filelog is not None:
                                          result = orig(lfutil.standin(path), fileid, filelog)
                                      else:
                                          result = orig(lfutil.standin(path), fileid)
                                      olddata = result.data
                                      result.data = lambda: olddata() + b'\0'
                                  return result
                              extensions.wrapfunction(ctx, 'filectx', filectx)
                          return ctx
                      # Figure out the status of big files and insert them into the
                      # appropriate list in the result. Also removes standin files
                      # from the listing. Revert to the original status if
                      # self.lfstatus is False.
                      # XXX large file status is buggy when used on repo proxy.
                      # XXX this needs to be investigated.
                      @localrepo.unfilteredmethod
                      def status(
                          self,
                          node1=b'.',
                          node2=None,
                          match=None,
                          ignored=False,
                          clean=False,
                          unknown=False,
                          listsubrepos=False,
                      ):
                          listignored, listclean, listunknown = ignored, clean, unknown
                          orig = super(lfilesrepo, self).status
                          if not self.lfstatus:
                              return orig(
                                  node1,
                                  node2,
                                  match,
                                  listignored,
                                  listclean,
                                  listunknown,
                                  listsubrepos,
                              )
                          # some calls in this function rely on the old version of status
                          self.lfstatus = False
                          ctx1 = self[node1]
                          ctx2 = self[node2]
                          working = ctx2.rev() is None
                          parentworking = working and ctx1 == self[b'.']
                          if match is None:
                              match = matchmod.always()
                          try:
                              # updating the dirstate is optional
                              # so we don't wait on the lock
                              wlock = self.wlock(False)
                              gotlock = True
                          except error.LockError:
                              wlock = util.nullcontextmanager()
                              gotlock = False
                          with wlock, self.dirstate.running_status(self):
                              # First check if paths or patterns were specified on the
                              # command line.  If there were, and they don't match any
                              # largefiles, we should just bail here and let super
                              # handle it -- thus gaining a big performance boost.
                              lfdirstate = lfutil.openlfdirstate(ui, self)
                              if not match.always():
                                  for f in lfdirstate:
                                      if match(f):
                                          break
                                  else:
                                      return orig(
                                          node1,
                                          node2,
                                          match,
                                          listignored,
                                          listclean,
                                          listunknown,
                                          listsubrepos,
                                      )
                              # Create a copy of match that matches standins instead
                              # of largefiles.
                              def tostandins(files):
                                  if not working:
                                      return files
                                  newfiles = []
                                  dirstate = self.dirstate
                                  for f in files:
                                      sf = lfutil.standin(f)
                                      if sf in dirstate:
                                          newfiles.append(sf)
                                      elif dirstate.hasdir(sf):
                                          # Directory entries could be regular or
                                          # standin, check both
                                          newfiles.extend((f, sf))
                                      else:
                                          newfiles.append(f)
                                  return newfiles
                              m = copy.copy(match)
                              m._files = tostandins(m._files)
                              result = orig(
                                  node1, node2, m, ignored, clean, unknown, listsubrepos
                              )
                              if working:
                                  def sfindirstate(f):
                                      sf = lfutil.standin(f)
                                      dirstate = self.dirstate
                                      return sf in dirstate or dirstate.hasdir(sf)
                                  match._files = [f for f in match._files if sfindirstate(f)]
                                  # Don't waste time getting the ignored and unknown
                                  # files from lfdirstate
                                  unsure, s, mtime_boundary = lfdirstate.status(
                                      match,
                                      subrepos=[],
                                      ignored=False,
                                      clean=listclean,
                                      unknown=False,
                                  )
                                  (modified, added, removed, deleted, clean) = (
                                      s.modified,
                                      s.added,
                                      s.removed,
                                      s.deleted,
                                      s.clean,
                                  )
                                  if parentworking:
                                      wctx = repo[None]
                                      for lfile in unsure:
                                          standin = lfutil.standin(lfile)
                                          if standin not in ctx1:
                                              # from second parent
                                              modified.append(lfile)
                                          elif lfutil.readasstandin(
                                              ctx1[standin]
                                          ) != lfutil.hashfile(self.wjoin(lfile)):
                                              modified.append(lfile)
                                          else:
                                              if listclean:
                                                  clean.append(lfile)
                                              s = wctx[lfile].lstat()
                                              mode = s.st_mode
                                              size = s.st_size
                                              mtime = timestamp.reliable_mtime_of(
                                                  s, mtime_boundary
                                              )
                                              if mtime is not None:
                                                  cache_data = (mode, size, mtime)
                                                  lfdirstate.set_clean(lfile, cache_data)
                                  else:
                                      tocheck = unsure + modified + added + clean
                                      modified, added, clean = [], [], []
                                      checkexec = self.dirstate._checkexec
                                      for lfile in tocheck:
                                          standin = lfutil.standin(lfile)
                                          if standin in ctx1:
                                              abslfile = self.wjoin(lfile)
                                              if (
                                                  lfutil.readasstandin(ctx1[standin])
                                                  != lfutil.hashfile(abslfile)
                                              ) or (
                                                  checkexec
                                                  and (b'x' in ctx1.flags(standin))
                                                  != bool(lfutil.getexecutable(abslfile))
                                              ):
                                                  modified.append(lfile)
                                              elif listclean:
                                                  clean.append(lfile)
                                          else:
                                              added.append(lfile)
                                      # at this point, 'removed' contains largefiles
                                      # marked as 'R' in the working context.
                                      # then, largefiles not managed also in the target
                                      # context should be excluded from 'removed'.
                                      removed = [
                                          lfile
                                          for lfile in removed
                                          if lfutil.standin(lfile) in ctx1
                                      ]
                                  # Standins no longer found in lfdirstate have been deleted
                                  for standin in ctx1.walk(lfutil.getstandinmatcher(self)):
                                      lfile = lfutil.splitstandin(standin)
                                      if not match(lfile):
                                          continue
                                      if lfile not in lfdirstate:
                                          deleted.append(lfile)
                                          # Sync "largefile has been removed" back to the
                                          # standin. Removing a file as a side effect of
                                          # running status is gross, but the alternatives (if
                                          # any) are worse.
                                          self.wvfs.unlinkpath(standin, ignoremissing=True)
                                  # Filter result lists
                                  result = list(result)
                                  # Largefiles are not really removed when they're
                                  # still in the normal dirstate. Likewise, normal
                                  # files are not really removed if they are still in
                                  # lfdirstate. This happens in merges where files
                                  # change type.
                                  removed = [f for f in removed if f not in self.dirstate]
                                  result[2] = [f for f in result[2] if f not in lfdirstate]
                                  lfiles = set(lfdirstate)
                                  # Unknown files
                                  result[4] = set(result[4]).difference(lfiles)
                                  # Ignored files
                                  result[5] = set(result[5]).difference(lfiles)
                                  # combine normal files and largefiles
                                  normals = [
                                      [fn for fn in filelist if not lfutil.isstandin(fn)]
                                      for filelist in result
                                  ]
                                  lfstatus = (
                                      modified,
                                      added,
                                      removed,
                                      deleted,
                                      [],
                                      [],
                                      clean,
                                  )
                                  result = [
                                      sorted(list1 + list2)
                                      for (list1, list2) in zip(normals, lfstatus)
                                  ]
                              else:  # not against working directory
                                  result = [
                                      [lfutil.splitstandin(f) or f for f in items]
                                      for items in result
                                  ]
                              if gotlock:
                                  lfdirstate.write(self.currenttransaction())
                              else:
                                  lfdirstate.invalidate()
                          self.lfstatus = True
                          return scmutil.status(*result)
                      def commitctx(self, ctx, *args, **kwargs):
                          node = super(lfilesrepo, self).commitctx(ctx, *args, **kwargs)
                          class lfilesctx(ctx.__class__):
                              def markcommitted(self, node):
                                  orig = super(lfilesctx, self).markcommitted
                                  return lfutil.markcommitted(orig, self, node)
                          ctx.__class__ = lfilesctx
                          return node
                      # Before commit, largefile standins have not had their
                      # contents updated to reflect the hash of their largefile.
                      # Do that here.
                      def commit(
                          self,
                          text=b"",
                          user=None,
                          date=None,
                          match=None,
                          force=False,
                          editor=False,
                          extra=None,
                      ):
                          if extra is None:
                              extra = {}
                          orig = super(lfilesrepo, self).commit
                          with self.wlock():
                              lfcommithook = self._lfcommithooks[-1]
                              match = lfcommithook(self, match)
                              result = orig(
                                  text=text,
                                  user=user,
                                  date=date,
                                  match=match,
                                  force=force,
                                  editor=editor,
                                  extra=extra,
                              )
                              return result
                      # TODO: _subdirlfs should be moved into "lfutil.py", because
                      # it is referred only from "lfutil.updatestandinsbymatch"
                      def _subdirlfs(self, files, lfiles):
                          """
                          Adjust matched file list
                          If we pass a directory to commit whose only committable files
                          are largefiles, the core commit code aborts before finding
                          the largefiles.
                          So we do the following:
                          For directories that only have largefiles as matches,
                          we explicitly add the largefiles to the match list and remove
                          the directory.
                          In other cases, we leave the match list unmodified.
                          """
                          actualfiles = []
                          dirs = []
                          regulars = []
                          for f in files:
                              if lfutil.isstandin(f + b'/'):
                                  raise error.Abort(
                                      _(b'file "%s" is a largefile standin') % f,
                                      hint=b'commit the largefile itself instead',
                                  )
                              # Scan directories
                              if self.wvfs.isdir(f):
                                  dirs.append(f)
                              else:
                                  regulars.append(f)
                          for f in dirs:
                              matcheddir = False
                              d = self.dirstate.normalize(f) + b'/'
                              # Check for matched normal files
                              for mf in regulars:
                                  if self.dirstate.normalize(mf).startswith(d):
                                      actualfiles.append(f)
                                      matcheddir = True
                                      break
                              if not matcheddir:
                                  # If no normal match, manually append
                                  # any matching largefiles
                                  for lf in lfiles:
                                      if self.dirstate.normalize(lf).startswith(d):
                                          actualfiles.append(lf)
                                          if not matcheddir:
                                              # There may still be normal files in the dir, so
                                              # add a directory to the list, which
                                              # forces status/dirstate to walk all files and
                                              # call the match function on the matcher, even
                                              # on case sensitive filesystems.
                                              actualfiles.append(b'.')
                                              matcheddir = True
                              # Nothing in dir, so readd it
                              # and let commit reject it
                              if not matcheddir:
                                  actualfiles.append(f)
                          # Always add normal files
                          actualfiles += regulars
                          return actualfiles
                  repo.__class__ = lfilesrepo
                  # stack of hooks being executed before committing.
                  # only last element ("_lfcommithooks[-1]") is used for each committing.
                  repo._lfcommithooks = [lfutil.updatestandinsbymatch]
                  # Stack of status writer functions taking "*msg, **opts" arguments
                  # like "ui.status()". Only last element ("_lfstatuswriters[-1]")
                  # is used to write status out.
                  repo._lfstatuswriters = [ui.status]
                  def prepushoutgoinghook(pushop):
                      """Push largefiles for pushop before pushing revisions."""
                      lfrevs = pushop.lfrevs
                      if lfrevs is None:
                          lfrevs = pushop.outgoing.missing
                      if lfrevs:
                          toupload = set()
                          addfunc = lambda fn, lfhash: toupload.add(lfhash)
                          lfutil.getlfilestoupload(pushop.repo, lfrevs, addfunc)
                          lfcommands.uploadlfiles(ui, pushop.repo, pushop.remote, toupload)
                  repo.prepushoutgoinghooks.add(b"largefiles", prepushoutgoinghook)
                  def checkrequireslfiles(ui, repo, **kwargs):
                      with repo.lock():
                          if b'largefiles' in repo.requirements:
                              return
                          marker = lfutil.shortnameslash
-                         for entry in repo.store.datafiles():
+                         for entry in repo.store.data_entries():
                              # XXX note that this match is not rooted and can wrongly match
                              # directory ending with ".hglf"
                              if entry.is_revlog and marker in entry.target_id:
                                  repo.requirements.add(b'largefiles')
                                  scmutil.writereporequirements(repo)
                                  break
                  ui.setconfig(
                      b'hooks', b'changegroup.lfiles', checkrequireslfiles, b'largefiles'
                  )
                  ui.setconfig(b'hooks', b'commit.lfiles', checkrequireslfiles, b'largefiles')

hgext/narrow/narrowcommands.py

0 +1 -1

              # narrowcommands.py - command modifications for narrowhg extension
              #
              # Copyright 2017 Google, Inc.
              #
              # This software may be used and distributed according to the terms of the
              # GNU General Public License version 2 or any later version.
              import itertools
              import os
              from mercurial.i18n import _
              from mercurial.node import (
                  hex,
                  short,
              )
              from mercurial import (
                  bundle2,
                  cmdutil,
                  commands,
                  discovery,
                  encoding,
                  error,
                  exchange,
                  extensions,
                  hg,
                  narrowspec,
                  pathutil,
                  pycompat,
                  registrar,
                  repair,
                  repoview,
                  requirements,
                  sparse,
                  util,
                  wireprototypes,
              )
              from mercurial.utils import (
                  urlutil,
              )
              table = {}
              command = registrar.command(table)
              def setup():
                  """Wraps user-facing mercurial commands with narrow-aware versions."""
                  entry = extensions.wrapcommand(commands.table, b'clone', clonenarrowcmd)
                  entry[1].append(
                      (b'', b'narrow', None, _(b"create a narrow clone of select files"))
                  )
                  entry[1].append(
                      (
                          b'',
                          b'depth',
                          b'',
                          _(b"limit the history fetched by distance from heads"),
                      )
                  )
                  entry[1].append((b'', b'narrowspec', b'', _(b"read narrowspecs from file")))
                  # TODO(durin42): unify sparse/narrow --include/--exclude logic a bit
                  if b'sparse' not in extensions.enabled():
                      entry[1].append(
                          (b'', b'include', [], _(b"specifically fetch this file/directory"))
                      )
                      entry[1].append(
                          (
                              b'',
                              b'exclude',
                              [],
                              _(b"do not fetch this file/directory, even if included"),
                          )
                      )
                  entry = extensions.wrapcommand(commands.table, b'pull', pullnarrowcmd)
                  entry[1].append(
                      (
                          b'',
                          b'depth',
                          b'',
                          _(b"limit the history fetched by distance from heads"),
                      )
                  )
                  extensions.wrapcommand(commands.table, b'archive', archivenarrowcmd)
              def clonenarrowcmd(orig, ui, repo, *args, **opts):
                  """Wraps clone command, so 'hg clone' first wraps localrepo.clone()."""
                  opts = pycompat.byteskwargs(opts)
                  wrappedextraprepare = util.nullcontextmanager()
                  narrowspecfile = opts[b'narrowspec']
                  if narrowspecfile:
                      filepath = os.path.join(encoding.getcwd(), narrowspecfile)
                      ui.status(_(b"reading narrowspec from '%s'\n") % filepath)
                      try:
                          fdata = util.readfile(filepath)
                      except IOError as inst:
                          raise error.Abort(
                              _(b"cannot read narrowspecs from '%s': %s")
                              % (filepath, encoding.strtolocal(inst.strerror))
                          )
                      includes, excludes, profiles = sparse.parseconfig(ui, fdata, b'narrow')
                      if profiles:
                          raise error.ConfigError(
                              _(
                                  b"cannot specify other files using '%include' in"
                                  b" narrowspec"
                              )
                          )
                      narrowspec.validatepatterns(includes)
                      narrowspec.validatepatterns(excludes)
                      # narrowspec is passed so we should assume that user wants narrow clone
                      opts[b'narrow'] = True
                      opts[b'include'].extend(includes)
                      opts[b'exclude'].extend(excludes)
                  if opts[b'narrow']:
                      def pullbundle2extraprepare_widen(orig, pullop, kwargs):
                          orig(pullop, kwargs)
                          if opts.get(b'depth'):
                              kwargs[b'depth'] = opts[b'depth']
                      wrappedextraprepare = extensions.wrappedfunction(
                          exchange, b'_pullbundle2extraprepare', pullbundle2extraprepare_widen
                      )
                  with wrappedextraprepare:
                      return orig(ui, repo, *args, **pycompat.strkwargs(opts))
              def pullnarrowcmd(orig, ui, repo, *args, **opts):
                  """Wraps pull command to allow modifying narrow spec."""
                  wrappedextraprepare = util.nullcontextmanager()
                  if requirements.NARROW_REQUIREMENT in repo.requirements:
                      def pullbundle2extraprepare_widen(orig, pullop, kwargs):
                          orig(pullop, kwargs)
                          if opts.get('depth'):
                              kwargs[b'depth'] = opts['depth']
                      wrappedextraprepare = extensions.wrappedfunction(
                          exchange, b'_pullbundle2extraprepare', pullbundle2extraprepare_widen
                      )
                  with wrappedextraprepare:
                      return orig(ui, repo, *args, **opts)
              def archivenarrowcmd(orig, ui, repo, *args, **opts):
                  """Wraps archive command to narrow the default includes."""
                  if requirements.NARROW_REQUIREMENT in repo.requirements:
                      repo_includes, repo_excludes = repo.narrowpats
                      includes = set(opts.get('include', []))
                      excludes = set(opts.get('exclude', []))
                      includes, excludes, unused_invalid = narrowspec.restrictpatterns(
                          includes, excludes, repo_includes, repo_excludes
                      )
                      if includes:
                          opts['include'] = includes
                      if excludes:
                          opts['exclude'] = excludes
                  return orig(ui, repo, *args, **opts)
              def pullbundle2extraprepare(orig, pullop, kwargs):
                  repo = pullop.repo
                  if requirements.NARROW_REQUIREMENT not in repo.requirements:
                      return orig(pullop, kwargs)
                  if wireprototypes.NARROWCAP not in pullop.remote.capabilities():
                      raise error.Abort(_(b"server does not support narrow clones"))
                  orig(pullop, kwargs)
                  kwargs[b'narrow'] = True
                  include, exclude = repo.narrowpats
                  kwargs[b'oldincludepats'] = include
                  kwargs[b'oldexcludepats'] = exclude
                  if include:
                      kwargs[b'includepats'] = include
                  if exclude:
                      kwargs[b'excludepats'] = exclude
                  # calculate known nodes only in ellipses cases because in non-ellipses cases
                  # we have all the nodes
                  if wireprototypes.ELLIPSESCAP1 in pullop.remote.capabilities():
                      kwargs[b'known'] = [
                          hex(ctx.node())
                          for ctx in repo.set(b'::%ln', pullop.common)
                          if ctx.node() != repo.nullid
                      ]
                      if not kwargs[b'known']:
                          # Mercurial serializes an empty list as '' and deserializes it as
                          # [''], so delete it instead to avoid handling the empty string on
                          # the server.
                          del kwargs[b'known']
              extensions.wrapfunction(
                  exchange, b'_pullbundle2extraprepare', pullbundle2extraprepare
              )
              def _narrow(
                  ui,
                  repo,
                  remote,
                  commoninc,
                  oldincludes,
                  oldexcludes,
                  newincludes,
                  newexcludes,
                  force,
                  backup,
              ):
                  oldmatch = narrowspec.match(repo.root, oldincludes, oldexcludes)
                  newmatch = narrowspec.match(repo.root, newincludes, newexcludes)
                  # This is essentially doing "hg outgoing" to find all local-only
                  # commits. We will then check that the local-only commits don't
                  # have any changes to files that will be untracked.
                  unfi = repo.unfiltered()
                  outgoing = discovery.findcommonoutgoing(unfi, remote, commoninc=commoninc)
                  ui.status(_(b'looking for local changes to affected paths\n'))
                  progress = ui.makeprogress(
                      topic=_(b'changesets'),
                      unit=_(b'changesets'),
                      total=len(outgoing.missing) + len(outgoing.excluded),
                  )
                  localnodes = []
                  with progress:
                      for n in itertools.chain(outgoing.missing, outgoing.excluded):
                          progress.increment()
                          if any(oldmatch(f) and not newmatch(f) for f in unfi[n].files()):
                              localnodes.append(n)
                  revstostrip = unfi.revs(b'descendants(%ln)', localnodes)
                  hiddenrevs = repoview.filterrevs(repo, b'visible')
                  visibletostrip = list(
                      repo.changelog.node(r) for r in (revstostrip - hiddenrevs)
                  )
                  if visibletostrip:
                      ui.status(
                          _(
                              b'The following changeset(s) or their ancestors have '
                              b'local changes not on the remote:\n'
                          )
                      )
                      maxnodes = 10
                      if ui.verbose or len(visibletostrip) <= maxnodes:
                          for n in visibletostrip:
                              ui.status(b'%s\n' % short(n))
                      else:
                          for n in visibletostrip[:maxnodes]:
                              ui.status(b'%s\n' % short(n))
                          ui.status(
                              _(b'...and %d more, use --verbose to list all\n')
                              % (len(visibletostrip) - maxnodes)
                          )
                      if not force:
                          raise error.StateError(
                              _(b'local changes found'),
                              hint=_(b'use --force-delete-local-changes to ignore'),
                          )
                  with ui.uninterruptible():
                      if revstostrip:
                          tostrip = [unfi.changelog.node(r) for r in revstostrip]
                          if repo[b'.'].node() in tostrip:
                              # stripping working copy, so move to a different commit first
                              urev = max(
                                  repo.revs(
                                      b'(::%n) - %ln + null',
                                      repo[b'.'].node(),
                                      visibletostrip,
                                  )
                              )
                              hg.clean(repo, urev)
                          overrides = {(b'devel', b'strip-obsmarkers'): False}
                          if backup:
                              ui.status(_(b'moving unwanted changesets to backup\n'))
                          else:
                              ui.status(_(b'deleting unwanted changesets\n'))
                          with ui.configoverride(overrides, b'narrow'):
                              repair.strip(ui, unfi, tostrip, topic=b'narrow', backup=backup)
                      todelete = []
-                     for entry in repo.store.datafiles():
+                     for entry in repo.store.data_entries():
                          if not entry.is_revlog:
                              continue
                          if entry.is_filelog:
                              if not newmatch(entry.target_id):
                                  for file_ in entry.files():
                                      todelete.append(file_.unencoded_path)
                          elif entry.is_manifestlog:
                              dir = entry.target_id
                              dirs = sorted(pathutil.dirs({dir})) + [dir]
                              include = True
                              for d in dirs:
                                  visit = newmatch.visitdir(d)
                                  if not visit:
                                      include = False
                                      break
                                  if visit == b'all':
                                      break
                              if not include:
                                  for file_ in entry.files():
                                      todelete.append(file_.unencoded_path)
                      repo.destroying()
                      with repo.transaction(b'narrowing'):
                          # Update narrowspec before removing revlogs, so repo won't be
                          # corrupt in case of crash
                          repo.setnarrowpats(newincludes, newexcludes)
                          for f in todelete:
                              ui.status(_(b'deleting %s\n') % f)
                              util.unlinkpath(repo.svfs.join(f))
                              repo.store.markremoved(f)
                          ui.status(_(b'deleting unwanted files from working copy\n'))
                          with repo.dirstate.changing_parents(repo):
                              narrowspec.updateworkingcopy(repo, assumeclean=True)
                              narrowspec.copytoworkingcopy(repo)
                      repo.destroyed()
              def _widen(
                  ui,
                  repo,
                  remote,
                  commoninc,
                  oldincludes,
                  oldexcludes,
                  newincludes,
                  newexcludes,
              ):
                  # for now we assume that if a server has ellipses enabled, we will be
                  # exchanging ellipses nodes. In future we should add ellipses as a client
                  # side requirement (maybe) to distinguish a client is shallow or not and
                  # then send that information to server whether we want ellipses or not.
                  # Theoretically a non-ellipses repo should be able to use narrow
                  # functionality from an ellipses enabled server
                  remotecap = remote.capabilities()
                  ellipsesremote = any(
                      cap in remotecap for cap in wireprototypes.SUPPORTED_ELLIPSESCAP
                  )
                  # check whether we are talking to a server which supports old version of
                  # ellipses capabilities
                  isoldellipses = (
                      ellipsesremote
                      and wireprototypes.ELLIPSESCAP1 in remotecap
                      and wireprototypes.ELLIPSESCAP not in remotecap
                  )
                  def pullbundle2extraprepare_widen(orig, pullop, kwargs):
                      orig(pullop, kwargs)
                      # The old{in,ex}cludepats have already been set by orig()
                      kwargs[b'includepats'] = newincludes
                      kwargs[b'excludepats'] = newexcludes
                  wrappedextraprepare = extensions.wrappedfunction(
                      exchange, b'_pullbundle2extraprepare', pullbundle2extraprepare_widen
                  )
                  # define a function that narrowbundle2 can call after creating the
                  # backup bundle, but before applying the bundle from the server
                  def setnewnarrowpats():
                      repo.setnarrowpats(newincludes, newexcludes)
                  repo.setnewnarrowpats = setnewnarrowpats
                  # silence the devel-warning of applying an empty changegroup
                  overrides = {(b'devel', b'all-warnings'): False}
                  common = commoninc[0]
                  with ui.uninterruptible():
                      if ellipsesremote:
                          ds = repo.dirstate
                          p1, p2 = ds.p1(), ds.p2()
                          with ds.changing_parents(repo):
                              ds.setparents(repo.nullid, repo.nullid)
                      if isoldellipses:
                          with wrappedextraprepare:
                              exchange.pull(repo, remote, heads=common)
                      else:
                          known = []
                          if ellipsesremote:
                              known = [
                                  ctx.node()
                                  for ctx in repo.set(b'::%ln', common)
                                  if ctx.node() != repo.nullid
                              ]
                          with remote.commandexecutor() as e:
                              bundle = e.callcommand(
                                  b'narrow_widen',
                                  {
                                      b'oldincludes': oldincludes,
                                      b'oldexcludes': oldexcludes,
                                      b'newincludes': newincludes,
                                      b'newexcludes': newexcludes,
                                      b'cgversion': b'03',
                                      b'commonheads': common,
                                      b'known': known,
                                      b'ellipses': ellipsesremote,
                                  },
                              ).result()
                          trmanager = exchange.transactionmanager(
                              repo, b'widen', remote.url()
                          )
                          with trmanager, repo.ui.configoverride(overrides, b'widen'):
                              op = bundle2.bundleoperation(
                                  repo, trmanager.transaction, source=b'widen'
                              )
                              # TODO: we should catch error.Abort here
                              bundle2.processbundle(repo, bundle, op=op, remote=remote)
                      if ellipsesremote:
                          with ds.changing_parents(repo):
                              ds.setparents(p1, p2)
                      with repo.transaction(b'widening'), repo.dirstate.changing_parents(
                          repo
                      ):
                          repo.setnewnarrowpats()
                          narrowspec.updateworkingcopy(repo)
                          narrowspec.copytoworkingcopy(repo)
              # TODO(rdamazio): Make new matcher format and update description
              @command(
                  b'tracked',
                  [
                      (b'', b'addinclude', [], _(b'new paths to include')),
                      (b'', b'removeinclude', [], _(b'old paths to no longer include')),
                      (
                          b'',
                          b'auto-remove-includes',
                          False,
                          _(b'automatically choose unused includes to remove'),
                      ),
                      (b'', b'addexclude', [], _(b'new paths to exclude')),
                      (b'', b'import-rules', b'', _(b'import narrowspecs from a file')),
                      (b'', b'removeexclude', [], _(b'old paths to no longer exclude')),
                      (
                          b'',
                          b'clear',
                          False,
                          _(b'whether to replace the existing narrowspec'),
                      ),
                      (
                          b'',
                          b'force-delete-local-changes',
                          False,
                          _(b'forces deletion of local changes when narrowing'),
                      ),
                      (
                          b'',
                          b'backup',
                          True,
                          _(b'back up local changes when narrowing'),
                      ),
                      (
                          b'',
                          b'update-working-copy',
                          False,
                          _(b'update working copy when the store has changed'),
                      ),
                  ]
                  + commands.remoteopts,
                  _(b'[OPTIONS]... [REMOTE]'),
                  inferrepo=True,
                  helpcategory=command.CATEGORY_MAINTENANCE,
              )
              def trackedcmd(ui, repo, remotepath=None, *pats, **opts):
                  """show or change the current narrowspec
                  With no argument, shows the current narrowspec entries, one per line. Each
                  line will be prefixed with 'I' or 'X' for included or excluded patterns,
                  respectively.
                  The narrowspec is comprised of expressions to match remote files and/or
                  directories that should be pulled into your client.
                  The narrowspec has *include* and *exclude* expressions, with excludes always
                  trumping includes: that is, if a file matches an exclude expression, it will
                  be excluded even if it also matches an include expression.
                  Excluding files that were never included has no effect.
                  Each included or excluded entry is in the format described by
                  'hg help patterns'.
                  The options allow you to add or remove included and excluded expressions.
                  If --clear is specified, then all previous includes and excludes are DROPPED
                  and replaced by the new ones specified to --addinclude and --addexclude.
                  If --clear is specified without any further options, the narrowspec will be
                  empty and will not match any files.
                  If --auto-remove-includes is specified, then those includes that don't match
                  any files modified by currently visible local commits (those not shared by
                  the remote) will be added to the set of explicitly specified includes to
                  remove.
                  --import-rules accepts a path to a file containing rules, allowing you to
                  add --addinclude, --addexclude rules in bulk. Like the other include and
                  exclude switches, the changes are applied immediately.
                  """
                  opts = pycompat.byteskwargs(opts)
                  if requirements.NARROW_REQUIREMENT not in repo.requirements:
                      raise error.InputError(
                          _(
                              b'the tracked command is only supported on '
                              b'repositories cloned with --narrow'
                          )
                      )
                  # Before supporting, decide whether it "hg tracked --clear" should mean
                  # tracking no paths or all paths.
                  if opts[b'clear']:
                      raise error.InputError(_(b'the --clear option is not yet supported'))
                  # import rules from a file
                  newrules = opts.get(b'import_rules')
                  if newrules:
                      try:
                          filepath = os.path.join(encoding.getcwd(), newrules)
                          fdata = util.readfile(filepath)
                      except IOError as inst:
                          raise error.StorageError(
                              _(b"cannot read narrowspecs from '%s': %s")
                              % (filepath, encoding.strtolocal(inst.strerror))
                          )
                      includepats, excludepats, profiles = sparse.parseconfig(
                          ui, fdata, b'narrow'
                      )
                      if profiles:
                          raise error.InputError(
                              _(
                                  b"including other spec files using '%include' "
                                  b"is not supported in narrowspec"
                              )
                          )
                      opts[b'addinclude'].extend(includepats)
                      opts[b'addexclude'].extend(excludepats)
                  addedincludes = narrowspec.parsepatterns(opts[b'addinclude'])
                  removedincludes = narrowspec.parsepatterns(opts[b'removeinclude'])
                  addedexcludes = narrowspec.parsepatterns(opts[b'addexclude'])
                  removedexcludes = narrowspec.parsepatterns(opts[b'removeexclude'])
                  autoremoveincludes = opts[b'auto_remove_includes']
                  update_working_copy = opts[b'update_working_copy']
                  only_show = not (
                      addedincludes
                      or removedincludes
                      or addedexcludes
                      or removedexcludes
                      or newrules
                      or autoremoveincludes
                      or update_working_copy
                  )
                  # Only print the current narrowspec.
                  if only_show:
                      oldincludes, oldexcludes = repo.narrowpats
                      ui.pager(b'tracked')
                      fm = ui.formatter(b'narrow', opts)
                      for i in sorted(oldincludes):
                          fm.startitem()
                          fm.write(b'status', b'%s ', b'I', label=b'narrow.included')
                          fm.write(b'pat', b'%s\n', i, label=b'narrow.included')
                      for i in sorted(oldexcludes):
                          fm.startitem()
                          fm.write(b'status', b'%s ', b'X', label=b'narrow.excluded')
                          fm.write(b'pat', b'%s\n', i, label=b'narrow.excluded')
                      fm.end()
                      return 0
                  with repo.wlock(), repo.lock():
                      oldincludes, oldexcludes = repo.narrowpats
                      # filter the user passed additions and deletions into actual additions and
                      # deletions of excludes and includes
                      addedincludes -= oldincludes
                      removedincludes &= oldincludes
                      addedexcludes -= oldexcludes
                      removedexcludes &= oldexcludes
                      widening = addedincludes or removedexcludes
                      narrowing = removedincludes or addedexcludes
                      if update_working_copy:
                          with repo.transaction(b'narrow-wc'), repo.dirstate.changing_parents(
                              repo
                          ):
                              narrowspec.updateworkingcopy(repo)
                              narrowspec.copytoworkingcopy(repo)
                          return 0
                      if not (widening or narrowing or autoremoveincludes):
                          ui.status(_(b"nothing to widen or narrow\n"))
                          return 0
                      cmdutil.bailifchanged(repo)
                      # Find the revisions we have in common with the remote. These will
                      # be used for finding local-only changes for narrowing. They will
                      # also define the set of revisions to update for widening.
                      path = urlutil.get_unique_pull_path_obj(b'tracked', ui, remotepath)
                      ui.status(_(b'comparing with %s\n') % urlutil.hidepassword(path.loc))
                      remote = hg.peer(repo, opts, path)
                      try:
                          # check narrow support before doing anything if widening needs to be
                          # performed. In future we should also abort if client is ellipses and
                          # server does not support ellipses
                          if (
                              widening
                              and wireprototypes.NARROWCAP not in remote.capabilities()
                          ):
                              raise error.Abort(_(b"server does not support narrow clones"))
                          commoninc = discovery.findcommonincoming(repo, remote)
                          if autoremoveincludes:
                              outgoing = discovery.findcommonoutgoing(
                                  repo, remote, commoninc=commoninc
                              )
                              ui.status(_(b'looking for unused includes to remove\n'))
                              localfiles = set()
                              for n in itertools.chain(outgoing.missing, outgoing.excluded):
                                  localfiles.update(repo[n].files())
                              suggestedremovals = []
                              for include in sorted(oldincludes):
                                  match = narrowspec.match(repo.root, [include], oldexcludes)
                                  if not any(match(f) for f in localfiles):
                                      suggestedremovals.append(include)
                              if suggestedremovals:
                                  for s in suggestedremovals:
                                      ui.status(b'%s\n' % s)
                                  if (
                                      ui.promptchoice(
                                          _(
                                              b'remove these unused includes (yn)?'
                                              b'$$ &Yes $$ &No'
                                          )
                                      )
                                      == 0
                                  ):
                                      removedincludes.update(suggestedremovals)
                                      narrowing = True
                              else:
                                  ui.status(_(b'found no unused includes\n'))
                          if narrowing:
                              newincludes = oldincludes - removedincludes
                              newexcludes = oldexcludes | addedexcludes
                              _narrow(
                                  ui,
                                  repo,
                                  remote,
                                  commoninc,
                                  oldincludes,
                                  oldexcludes,
                                  newincludes,
                                  newexcludes,
                                  opts[b'force_delete_local_changes'],
                                  opts[b'backup'],
                              )
                              # _narrow() updated the narrowspec and _widen() below needs to
                              # use the updated values as its base (otherwise removed includes
                              # and addedexcludes will be lost in the resulting narrowspec)
                              oldincludes = newincludes
                              oldexcludes = newexcludes
                          if widening:
                              newincludes = oldincludes | addedincludes
                              newexcludes = oldexcludes - removedexcludes
                              _widen(
                                  ui,
                                  repo,
                                  remote,
                                  commoninc,
                                  oldincludes,
                                  oldexcludes,
                                  newincludes,
                                  newexcludes,
                              )
                      finally:
                          remote.close()
                  return 0

hgext/remotefilelog/contentstore.py

0 +1 -1

              import threading
              from mercurial.node import (
                  hex,
                  sha1nodeconstants,
              )
              from mercurial.pycompat import getattr
              from mercurial import (
                  mdiff,
                  revlog,
              )
              from . import (
                  basestore,
                  constants,
                  shallowutil,
              )
              class ChainIndicies:
                  """A static class for easy reference to the delta chain indicies."""
                  # The filename of this revision delta
                  NAME = 0
                  # The mercurial file node for this revision delta
                  NODE = 1
                  # The filename of the delta base's revision. This is useful when delta
                  # between different files (like in the case of a move or copy, we can delta
                  # against the original file content).
                  BASENAME = 2
                  # The mercurial file node for the delta base revision. This is the nullid if
                  # this delta is a full text.
                  BASENODE = 3
                  # The actual delta or full text data.
                  DATA = 4
              class unioncontentstore(basestore.baseunionstore):
                  def __init__(self, *args, **kwargs):
                      super(unioncontentstore, self).__init__(*args, **kwargs)
                      self.stores = args
                      self.writestore = kwargs.get('writestore')
                      # If allowincomplete==True then the union store can return partial
                      # delta chains, otherwise it will throw a KeyError if a full
                      # deltachain can't be found.
                      self.allowincomplete = kwargs.get('allowincomplete', False)
                  def get(self, name, node):
                      """Fetches the full text revision contents of the given name+node pair.
                      If the full text doesn't exist, throws a KeyError.
                      Under the hood, this uses getdeltachain() across all the stores to build
                      up a full chain to produce the full text.
                      """
                      chain = self.getdeltachain(name, node)
                      if chain[-1][ChainIndicies.BASENODE] != sha1nodeconstants.nullid:
                          # If we didn't receive a full chain, throw
                          raise KeyError((name, hex(node)))
                      # The last entry in the chain is a full text, so we start our delta
                      # applies with that.
                      fulltext = chain.pop()[ChainIndicies.DATA]
                      text = fulltext
                      while chain:
                          delta = chain.pop()[ChainIndicies.DATA]
                          text = mdiff.patches(text, [delta])
                      return text
                  @basestore.baseunionstore.retriable
                  def getdelta(self, name, node):
                      """Return the single delta entry for the given name/node pair."""
                      for store in self.stores:
                          try:
                              return store.getdelta(name, node)
                          except KeyError:
                              pass
                      raise KeyError((name, hex(node)))
                  def getdeltachain(self, name, node):
                      """Returns the deltachain for the given name/node pair.
                      Returns an ordered list of:
                        [(name, node, deltabasename, deltabasenode, deltacontent),...]
                      where the chain is terminated by a full text entry with a nullid
                      deltabasenode.
                      """
                      chain = self._getpartialchain(name, node)
                      while chain[-1][ChainIndicies.BASENODE] != sha1nodeconstants.nullid:
                          x, x, deltabasename, deltabasenode, x = chain[-1]
                          try:
                              morechain = self._getpartialchain(deltabasename, deltabasenode)
                              chain.extend(morechain)
                          except KeyError:
                              # If we allow incomplete chains, don't throw.
                              if not self.allowincomplete:
                                  raise
                              break
                      return chain
                  @basestore.baseunionstore.retriable
                  def getmeta(self, name, node):
                      """Returns the metadata dict for given node."""
                      for store in self.stores:
                          try:
                              return store.getmeta(name, node)
                          except KeyError:
                              pass
                      raise KeyError((name, hex(node)))
                  def getmetrics(self):
                      metrics = [s.getmetrics() for s in self.stores]
                      return shallowutil.sumdicts(*metrics)
                  @basestore.baseunionstore.retriable
                  def _getpartialchain(self, name, node):
                      """Returns a partial delta chain for the given name/node pair.
                      A partial chain is a chain that may not be terminated in a full-text.
                      """
                      for store in self.stores:
                          try:
                              return store.getdeltachain(name, node)
                          except KeyError:
                              pass
                      raise KeyError((name, hex(node)))
                  def add(self, name, node, data):
                      raise RuntimeError(
                          b"cannot add content only to remotefilelog contentstore"
                      )
                  def getmissing(self, keys):
                      missing = keys
                      for store in self.stores:
                          if missing:
                              missing = store.getmissing(missing)
                      return missing
                  def addremotefilelognode(self, name, node, data):
                      if self.writestore:
                          self.writestore.addremotefilelognode(name, node, data)
                      else:
                          raise RuntimeError(b"no writable store configured")
                  def markledger(self, ledger, options=None):
                      for store in self.stores:
                          store.markledger(ledger, options)
              class remotefilelogcontentstore(basestore.basestore):
                  def __init__(self, *args, **kwargs):
                      super(remotefilelogcontentstore, self).__init__(*args, **kwargs)
                      self._threaddata = threading.local()
                  def get(self, name, node):
                      # return raw revision text
                      data = self._getdata(name, node)
                      offset, size, flags = shallowutil.parsesizeflags(data)
                      content = data[offset : offset + size]
                      ancestormap = shallowutil.ancestormap(data)
                      p1, p2, linknode, copyfrom = ancestormap[node]
                      copyrev = None
                      if copyfrom:
                          copyrev = hex(p1)
                      self._updatemetacache(node, size, flags)
                      # lfs tracks renames in its own metadata, remove hg copy metadata,
                      # because copy metadata will be re-added by lfs flag processor.
                      if flags & revlog.REVIDX_EXTSTORED:
                          copyrev = copyfrom = None
                      revision = shallowutil.createrevlogtext(content, copyfrom, copyrev)
                      return revision
                  def getdelta(self, name, node):
                      # Since remotefilelog content stores only contain full texts, just
                      # return that.
                      revision = self.get(name, node)
                      return (
                          revision,
                          name,
                          sha1nodeconstants.nullid,
                          self.getmeta(name, node),
                      )
                  def getdeltachain(self, name, node):
                      # Since remotefilelog content stores just contain full texts, we return
                      # a fake delta chain that just consists of a single full text revision.
                      # The nullid in the deltabasenode slot indicates that the revision is a
                      # fulltext.
                      revision = self.get(name, node)
                      return [(name, node, None, sha1nodeconstants.nullid, revision)]
                  def getmeta(self, name, node):
                      self._sanitizemetacache()
                      if node != self._threaddata.metacache[0]:
                          data = self._getdata(name, node)
                          offset, size, flags = shallowutil.parsesizeflags(data)
                          self._updatemetacache(node, size, flags)
                      return self._threaddata.metacache[1]
                  def add(self, name, node, data):
                      raise RuntimeError(
                          b"cannot add content only to remotefilelog contentstore"
                      )
                  def _sanitizemetacache(self):
                      metacache = getattr(self._threaddata, 'metacache', None)
                      if metacache is None:
                          self._threaddata.metacache = (None, None)  # (node, meta)
                  def _updatemetacache(self, node, size, flags):
                      self._sanitizemetacache()
                      if node == self._threaddata.metacache[0]:
                          return
                      meta = {constants.METAKEYFLAG: flags, constants.METAKEYSIZE: size}
                      self._threaddata.metacache = (node, meta)
              class remotecontentstore:
                  def __init__(self, ui, fileservice, shared):
                      self._fileservice = fileservice
                      # type(shared) is usually remotefilelogcontentstore
                      self._shared = shared
                  def get(self, name, node):
                      self._fileservice.prefetch(
                          [(name, hex(node))], force=True, fetchdata=True
                      )
                      return self._shared.get(name, node)
                  def getdelta(self, name, node):
                      revision = self.get(name, node)
                      return (
                          revision,
                          name,
                          sha1nodeconstants.nullid,
                          self._shared.getmeta(name, node),
                      )
                  def getdeltachain(self, name, node):
                      # Since our remote content stores just contain full texts, we return a
                      # fake delta chain that just consists of a single full text revision.
                      # The nullid in the deltabasenode slot indicates that the revision is a
                      # fulltext.
                      revision = self.get(name, node)
                      return [(name, node, None, sha1nodeconstants.nullid, revision)]
                  def getmeta(self, name, node):
                      self._fileservice.prefetch(
                          [(name, hex(node))], force=True, fetchdata=True
                      )
                      return self._shared.getmeta(name, node)
                  def add(self, name, node, data):
                      raise RuntimeError(b"cannot add to a remote store")
                  def getmissing(self, keys):
                      return keys
                  def markledger(self, ledger, options=None):
                      pass
              class manifestrevlogstore:
                  def __init__(self, repo):
                      self._store = repo.store
                      self._svfs = repo.svfs
                      self._revlogs = dict()
                      self._cl = revlog.revlog(self._svfs, radix=b'00changelog.i')
                      self._repackstartlinkrev = 0
                  def get(self, name, node):
                      return self._revlog(name).rawdata(node)
                  def getdelta(self, name, node):
                      revision = self.get(name, node)
                      return revision, name, self._cl.nullid, self.getmeta(name, node)
                  def getdeltachain(self, name, node):
                      revision = self.get(name, node)
                      return [(name, node, None, self._cl.nullid, revision)]
                  def getmeta(self, name, node):
                      rl = self._revlog(name)
                      rev = rl.rev(node)
                      return {
                          constants.METAKEYFLAG: rl.flags(rev),
                          constants.METAKEYSIZE: rl.rawsize(rev),
                      }
                  def getancestors(self, name, node, known=None):
                      if known is None:
                          known = set()
                      if node in known:
                          return []
                      rl = self._revlog(name)
                      ancestors = {}
                      missing = {node}
                      for ancrev in rl.ancestors([rl.rev(node)], inclusive=True):
                          ancnode = rl.node(ancrev)
                          missing.discard(ancnode)
                          p1, p2 = rl.parents(ancnode)
                          if p1 != self._cl.nullid and p1 not in known:
                              missing.add(p1)
                          if p2 != self._cl.nullid and p2 not in known:
                              missing.add(p2)
                          linknode = self._cl.node(rl.linkrev(ancrev))
                          ancestors[rl.node(ancrev)] = (p1, p2, linknode, b'')
                          if not missing:
                              break
                      return ancestors
                  def getnodeinfo(self, name, node):
                      cl = self._cl
                      rl = self._revlog(name)
                      parents = rl.parents(node)
                      linkrev = rl.linkrev(rl.rev(node))
                      return (parents[0], parents[1], cl.node(linkrev), None)
                  def add(self, *args):
                      raise RuntimeError(b"cannot add to a revlog store")
                  def _revlog(self, name):
                      rl = self._revlogs.get(name)
                      if rl is None:
                          revlogname = b'00manifesttree'
                          if name != b'':
                              revlogname = b'meta/%s/00manifest' % name
                          rl = revlog.revlog(self._svfs, radix=revlogname)
                          self._revlogs[name] = rl
                      return rl
                  def getmissing(self, keys):
                      missing = []
                      for name, node in keys:
                          mfrevlog = self._revlog(name)
                          if node not in mfrevlog.nodemap:
                              missing.append((name, node))
                      return missing
                  def setrepacklinkrevrange(self, startrev, endrev):
                      self._repackstartlinkrev = startrev
                      self._repackendlinkrev = endrev
                  def markledger(self, ledger, options=None):
                      if options and options.get(constants.OPTION_PACKSONLY):
                          return
                      treename = b''
                      rl = revlog.revlog(self._svfs, radix=b'00manifesttree')
                      startlinkrev = self._repackstartlinkrev
                      endlinkrev = self._repackendlinkrev
                      for rev in range(len(rl) - 1, -1, -1):
                          linkrev = rl.linkrev(rev)
                          if linkrev < startlinkrev:
                              break
                          if linkrev > endlinkrev:
                              continue
                          node = rl.node(rev)
                          ledger.markdataentry(self, treename, node)
                          ledger.markhistoryentry(self, treename, node)
-                     for t, path, size in self._store.datafiles():
+                     for t, path, size in self._store.data_entries():
                          if path[:5] != b'meta/' or path[-2:] != b'.i':
                              continue
                          treename = path[5 : -len(b'/00manifest')]
                          rl = revlog.revlog(self._svfs, indexfile=path[:-2])
                          for rev in range(len(rl) - 1, -1, -1):
                              linkrev = rl.linkrev(rev)
                              if linkrev < startlinkrev:
                                  break
                              if linkrev > endlinkrev:
                                  continue
                              node = rl.node(rev)
                              ledger.markdataentry(self, treename, node)
                              ledger.markhistoryentry(self, treename, node)
                  def cleanup(self, ledger):
                      pass

hgext/remotefilelog/remotefilelogserver.py

0 +2 -2

              # remotefilelogserver.py - server logic for a remotefilelog server
              #
              # Copyright 2013 Facebook, Inc.
              #
              # This software may be used and distributed according to the terms of the
              # GNU General Public License version 2 or any later version.
              import os
              import stat
              import time
              import zlib
              from mercurial.i18n import _
              from mercurial.node import bin, hex
              from mercurial.pycompat import open
              from mercurial import (
                  changegroup,
                  changelog,
                  context,
                  error,
                  extensions,
                  match,
                  scmutil,
                  store,
                  streamclone,
                  util,
                  wireprotoserver,
                  wireprototypes,
                  wireprotov1server,
              )
              from . import (
                  constants,
                  shallowutil,
              )
              _sshv1server = wireprotoserver.sshv1protocolhandler
              def setupserver(ui, repo):
                  """Sets up a normal Mercurial repo so it can serve files to shallow repos."""
                  onetimesetup(ui)
                  # don't send files to shallow clients during pulls
                  def generatefiles(
                      orig, self, changedfiles, linknodes, commonrevs, source, *args, **kwargs
                  ):
                      caps = self._bundlecaps or []
                      if constants.BUNDLE2_CAPABLITY in caps:
                          # only send files that don't match the specified patterns
                          includepattern = None
                          excludepattern = None
                          for cap in self._bundlecaps or []:
                              if cap.startswith(b"includepattern="):
                                  includepattern = cap[len(b"includepattern=") :].split(b'\0')
                              elif cap.startswith(b"excludepattern="):
                                  excludepattern = cap[len(b"excludepattern=") :].split(b'\0')
                          m = match.always()
                          if includepattern or excludepattern:
                              m = match.match(
                                  repo.root, b'', None, includepattern, excludepattern
                              )
                          changedfiles = list([f for f in changedfiles if not m(f)])
                      return orig(
                          self, changedfiles, linknodes, commonrevs, source, *args, **kwargs
                      )
                  extensions.wrapfunction(
                      changegroup.cgpacker, b'generatefiles', generatefiles
                  )
              onetime = False
              def onetimesetup(ui):
                  """Configures the wireprotocol for both clients and servers."""
                  global onetime
                  if onetime:
                      return
                  onetime = True
                  # support file content requests
                  wireprotov1server.wireprotocommand(
                      b'x_rfl_getflogheads', b'path', permission=b'pull'
                  )(getflogheads)
                  wireprotov1server.wireprotocommand(
                      b'x_rfl_getfiles', b'', permission=b'pull'
                  )(getfiles)
                  wireprotov1server.wireprotocommand(
                      b'x_rfl_getfile', b'file node', permission=b'pull'
                  )(getfile)
                  class streamstate:
                      match = None
                      shallowremote = False
                      noflatmf = False
                  state = streamstate()
                  def stream_out_shallow(repo, proto, other):
                      includepattern = None
                      excludepattern = None
                      raw = other.get(b'includepattern')
                      if raw:
                          includepattern = raw.split(b'\0')
                      raw = other.get(b'excludepattern')
                      if raw:
                          excludepattern = raw.split(b'\0')
                      oldshallow = state.shallowremote
                      oldmatch = state.match
                      oldnoflatmf = state.noflatmf
                      try:
                          state.shallowremote = True
                          state.match = match.always()
                          state.noflatmf = other.get(b'noflatmanifest') == b'True'
                          if includepattern or excludepattern:
                              state.match = match.match(
                                  repo.root, b'', None, includepattern, excludepattern
                              )
                          streamres = wireprotov1server.stream(repo, proto)
                          # Force the first value to execute, so the file list is computed
                          # within the try/finally scope
                          first = next(streamres.gen)
                          second = next(streamres.gen)
                          def gen():
                              yield first
                              yield second
                              for value in streamres.gen:
                                  yield value
                          return wireprototypes.streamres(gen())
                      finally:
                          state.shallowremote = oldshallow
                          state.match = oldmatch
                          state.noflatmf = oldnoflatmf
                  wireprotov1server.commands[b'stream_out_shallow'] = (
                      stream_out_shallow,
                      b'*',
                  )
                  # don't clone filelogs to shallow clients
                  def _walkstreamfiles(orig, repo, matcher=None):
                      if state.shallowremote:
                          # if we are shallow ourselves, stream our local commits
                          if shallowutil.isenabled(repo):
                              striplen = len(repo.store.path) + 1
                              readdir = repo.store.rawvfs.readdir
                              visit = [os.path.join(repo.store.path, b'data')]
                              while visit:
                                  p = visit.pop()
                                  for f, kind, st in readdir(p, stat=True):
                                      fp = p + b'/' + f
                                      if kind == stat.S_IFREG:
                                          if not fp.endswith(b'.i') and not fp.endswith(
                                              b'.d'
                                          ):
                                              n = util.pconvert(fp[striplen:])
                                              d = store.decodedir(n)
                                              yield store.SimpleStoreEntry(
                                                  entry_path=d,
                                                  is_volatile=False,
                                                  file_size=st.st_size,
                                              )
                                      if kind == stat.S_IFDIR:
                                          visit.append(fp)
                          if scmutil.istreemanifest(repo):
-                             for entry in repo.store.datafiles():
+                             for entry in repo.store.data_entries():
                                  if not entry.is_revlog:
                                      continue
                                  if entry.is_manifestlog:
                                      yield entry
                          # Return .d and .i files that do not match the shallow pattern
                          match = state.match
                          if match and not match.always():
-                             for entry in repo.store.datafiles():
+                             for entry in repo.store.data_entries():
                                  if not entry.is_revlog:
                                      continue
                                  if not state.match(entry.target_id):
                                      yield entry
                          for x in repo.store.topfiles():
                              if state.noflatmf and x[1][:11] == b'00manifest.':
                                  continue
                              yield x
                      elif shallowutil.isenabled(repo):
                          # don't allow cloning from a shallow repo to a full repo
                          # since it would require fetching every version of every
                          # file in order to create the revlogs.
                          raise error.Abort(
                              _(b"Cannot clone from a shallow repo to a full repo.")
                          )
                      else:
                          for x in orig(repo, matcher):
                              yield x
                  extensions.wrapfunction(streamclone, b'_walkstreamfiles', _walkstreamfiles)
                  # expose remotefilelog capabilities
                  def _capabilities(orig, repo, proto):
                      caps = orig(repo, proto)
                      if shallowutil.isenabled(repo) or ui.configbool(
                          b'remotefilelog', b'server'
                      ):
                          if isinstance(proto, _sshv1server):
                              # legacy getfiles method which only works over ssh
                              caps.append(constants.NETWORK_CAP_LEGACY_SSH_GETFILES)
                          caps.append(b'x_rfl_getflogheads')
                          caps.append(b'x_rfl_getfile')
                      return caps
                  extensions.wrapfunction(wireprotov1server, b'_capabilities', _capabilities)
                  def _adjustlinkrev(orig, self, *args, **kwargs):
                      # When generating file blobs, taking the real path is too slow on large
                      # repos, so force it to just return the linkrev directly.
                      repo = self._repo
                      if util.safehasattr(repo, b'forcelinkrev') and repo.forcelinkrev:
                          return self._filelog.linkrev(self._filelog.rev(self._filenode))
                      return orig(self, *args, **kwargs)
                  extensions.wrapfunction(
                      context.basefilectx, b'_adjustlinkrev', _adjustlinkrev
                  )
                  def _iscmd(orig, cmd):
                      if cmd == b'x_rfl_getfiles':
                          return False
                      return orig(cmd)
                  extensions.wrapfunction(wireprotoserver, b'iscmd', _iscmd)
              def _loadfileblob(repo, cachepath, path, node):
                  filecachepath = os.path.join(cachepath, path, hex(node))
                  if not os.path.exists(filecachepath) or os.path.getsize(filecachepath) == 0:
                      filectx = repo.filectx(path, fileid=node)
                      if filectx.node() == repo.nullid:
                          repo.changelog = changelog.changelog(repo.svfs)
                          filectx = repo.filectx(path, fileid=node)
                      text = createfileblob(filectx)
                      # TODO configurable compression engines
                      text = zlib.compress(text)
                      # everything should be user & group read/writable
                      oldumask = os.umask(0o002)
                      try:
                          dirname = os.path.dirname(filecachepath)
                          if not os.path.exists(dirname):
                              try:
                                  os.makedirs(dirname)
                              except FileExistsError:
                                  pass
                          f = None
                          try:
                              f = util.atomictempfile(filecachepath, b"wb")
                              f.write(text)
                          except (IOError, OSError):
                              # Don't abort if the user only has permission to read,
                              # and not write.
                              pass
                          finally:
                              if f:
                                  f.close()
                      finally:
                          os.umask(oldumask)
                  else:
                      with open(filecachepath, b"rb") as f:
                          text = f.read()
                  return text
              def getflogheads(repo, proto, path):
                  """A server api for requesting a filelog's heads"""
                  flog = repo.file(path)
                  heads = flog.heads()
                  return b'\n'.join((hex(head) for head in heads if head != repo.nullid))
              def getfile(repo, proto, file, node):
                  """A server api for requesting a particular version of a file. Can be used
                  in batches to request many files at once. The return protocol is:
                  <errorcode>\0<data/errormsg> where <errorcode> is 0 for success or
                  non-zero for an error.
                  data is a compressed blob with revlog flag and ancestors information. See
                  createfileblob for its content.
                  """
                  if shallowutil.isenabled(repo):
                      return b'1\0' + _(b'cannot fetch remote files from shallow repo')
                  cachepath = repo.ui.config(b"remotefilelog", b"servercachepath")
                  if not cachepath:
                      cachepath = os.path.join(repo.path, b"remotefilelogcache")
                  node = bin(node.strip())
                  if node == repo.nullid:
                      return b'0\0'
                  return b'0\0' + _loadfileblob(repo, cachepath, file, node)
              def getfiles(repo, proto):
                  """A server api for requesting particular versions of particular files."""
                  if shallowutil.isenabled(repo):
                      raise error.Abort(_(b'cannot fetch remote files from shallow repo'))
                  if not isinstance(proto, _sshv1server):
                      raise error.Abort(_(b'cannot fetch remote files over non-ssh protocol'))
                  def streamer():
                      fin = proto._fin
                      cachepath = repo.ui.config(b"remotefilelog", b"servercachepath")
                      if not cachepath:
                          cachepath = os.path.join(repo.path, b"remotefilelogcache")
                      while True:
                          request = fin.readline()[:-1]
                          if not request:
                              break
                          node = bin(request[:40])
                          if node == repo.nullid:
                              yield b'0\n'
                              continue
                          path = request[40:]
                          text = _loadfileblob(repo, cachepath, path, node)
                          yield b'%d\n%s' % (len(text), text)
                          # it would be better to only flush after processing a whole batch
                          # but currently we don't know if there are more requests coming
                          proto._fout.flush()
                  return wireprototypes.streamres(streamer())
              def createfileblob(filectx):
                  """
                  format:
                      v0:
                          str(len(rawtext)) + '\0' + rawtext + ancestortext
                      v1:
                          'v1' + '\n' + metalist + '\0' + rawtext + ancestortext
                          metalist := metalist + '\n' + meta | meta
                          meta := sizemeta | flagmeta
                          sizemeta := METAKEYSIZE + str(len(rawtext))
                          flagmeta := METAKEYFLAG + str(flag)
                          note: sizemeta must exist. METAKEYFLAG and METAKEYSIZE must have a
                          length of 1.
                  """
                  flog = filectx.filelog()
                  frev = filectx.filerev()
                  revlogflags = flog._revlog.flags(frev)
                  if revlogflags == 0:
                      # normal files
                      text = filectx.data()
                  else:
                      # lfs, read raw revision data
                      text = flog.rawdata(frev)
                  repo = filectx._repo
                  ancestors = [filectx]
                  try:
                      repo.forcelinkrev = True
                      ancestors.extend([f for f in filectx.ancestors()])
                      ancestortext = b""
                      for ancestorctx in ancestors:
                          parents = ancestorctx.parents()
                          p1 = repo.nullid
                          p2 = repo.nullid
                          if len(parents) > 0:
                              p1 = parents[0].filenode()
                          if len(parents) > 1:
                              p2 = parents[1].filenode()
                          copyname = b""
                          rename = ancestorctx.renamed()
                          if rename:
                              copyname = rename[0]
                          linknode = ancestorctx.node()
                          ancestortext += b"%s%s%s%s%s\0" % (
                              ancestorctx.filenode(),
                              p1,
                              p2,
                              linknode,
                              copyname,
                          )
                  finally:
                      repo.forcelinkrev = False
                  header = shallowutil.buildfileblobheader(len(text), revlogflags)
                  return b"%s\0%s%s" % (header, text, ancestortext)
              def gcserver(ui, repo):
                  if not repo.ui.configbool(b"remotefilelog", b"server"):
                      return
                  neededfiles = set()
                  heads = repo.revs(b"heads(tip~25000:) - null")
                  cachepath = repo.vfs.join(b"remotefilelogcache")
                  for head in heads:
                      mf = repo[head].manifest()
                      for filename, filenode in mf.items():
                          filecachepath = os.path.join(cachepath, filename, hex(filenode))
                          neededfiles.add(filecachepath)
                  # delete unneeded older files
                  days = repo.ui.configint(b"remotefilelog", b"serverexpiration")
                  expiration = time.time() - (days * 24 * 60 * 60)
                  progress = ui.makeprogress(_(b"removing old server cache"), unit=b"files")
                  progress.update(0)
                  for root, dirs, files in os.walk(cachepath):
                      for file in files:
                          filepath = os.path.join(root, file)
                          progress.increment()
                          if filepath in neededfiles:
                              continue
                          stat = os.stat(filepath)
                          if stat.st_mtime < expiration:
                              os.remove(filepath)
                  progress.complete()

mercurial/repair.py

0 +1 -1

              # repair.py - functions for repository repair for mercurial
              #
              # Copyright 2005, 2006 Chris Mason <mason@suse.com>
              # Copyright 2007 Olivia Mackall
              #
              # This software may be used and distributed according to the terms of the
              # GNU General Public License version 2 or any later version.
              from .i18n import _
              from .node import (
                  hex,
                  short,
              )
              from . import (
                  bundle2,
                  changegroup,
                  discovery,
                  error,
                  exchange,
                  obsolete,
                  obsutil,
                  pathutil,
                  phases,
                  requirements,
                  scmutil,
                  store,
                  transaction,
                  util,
              )
              from .utils import (
                  hashutil,
                  urlutil,
              )
              def backupbundle(
                  repo,
                  bases,
                  heads,
                  node,
                  suffix,
                  compress=True,
                  obsolescence=True,
                  tmp_backup=False,
              ):
                  """create a bundle with the specified revisions as a backup"""
                  backupdir = b"strip-backup"
                  vfs = repo.vfs
                  if not vfs.isdir(backupdir):
                      vfs.mkdir(backupdir)
                  # Include a hash of all the nodes in the filename for uniqueness
                  allcommits = repo.set(b'%ln::%ln', bases, heads)
                  allhashes = sorted(c.hex() for c in allcommits)
                  totalhash = hashutil.sha1(b''.join(allhashes)).digest()
                  name = b"%s/%s-%s-%s.hg" % (
                      backupdir,
                      short(node),
                      hex(totalhash[:4]),
                      suffix,
                  )
                  cgversion = changegroup.localversion(repo)
                  comp = None
                  if cgversion != b'01':
                      bundletype = b"HG20"
                      if compress:
                          comp = b'BZ'
                  elif compress:
                      bundletype = b"HG10BZ"
                  else:
                      bundletype = b"HG10UN"
                  outgoing = discovery.outgoing(repo, missingroots=bases, ancestorsof=heads)
                  contentopts = {
                      b'cg.version': cgversion,
                      b'obsolescence': obsolescence,
                      b'phases': True,
                  }
                  return bundle2.writenewbundle(
                      repo.ui,
                      repo,
                      b'strip',
                      name,
                      bundletype,
                      outgoing,
                      contentopts,
                      vfs,
                      compression=comp,
                      allow_internal=tmp_backup,
                  )
              def _collectfiles(repo, striprev):
                  """find out the filelogs affected by the strip"""
                  files = set()
                  for x in range(striprev, len(repo)):
                      files.update(repo[x].files())
                  return sorted(files)
              def _collectrevlog(revlog, striprev):
                  _, brokenset = revlog.getstrippoint(striprev)
                  return [revlog.linkrev(r) for r in brokenset]
              def _collectbrokencsets(repo, files, striprev):
                  """return the changesets which will be broken by the truncation"""
                  s = set()
                  for revlog in manifestrevlogs(repo):
                      s.update(_collectrevlog(revlog, striprev))
                  for fname in files:
                      s.update(_collectrevlog(repo.file(fname), striprev))
                  return s
              def strip(ui, repo, nodelist, backup=True, topic=b'backup'):
                  # This function requires the caller to lock the repo, but it operates
                  # within a transaction of its own, and thus requires there to be no current
                  # transaction when it is called.
                  if repo.currenttransaction() is not None:
                      raise error.ProgrammingError(b'cannot strip from inside a transaction')
                  # Simple way to maintain backwards compatibility for this
                  # argument.
                  if backup in [b'none', b'strip']:
                      backup = False
                  repo = repo.unfiltered()
                  repo.destroying()
                  vfs = repo.vfs
                  # load bookmark before changelog to avoid side effect from outdated
                  # changelog (see repo._refreshchangelog)
                  repo._bookmarks
                  cl = repo.changelog
                  # TODO handle undo of merge sets
                  if isinstance(nodelist, bytes):
                      nodelist = [nodelist]
                  striplist = [cl.rev(node) for node in nodelist]
                  striprev = min(striplist)
                  files = _collectfiles(repo, striprev)
                  saverevs = _collectbrokencsets(repo, files, striprev)
                  # Some revisions with rev > striprev may not be descendants of striprev.
                  # We have to find these revisions and put them in a bundle, so that
                  # we can restore them after the truncations.
                  # To create the bundle we use repo.changegroupsubset which requires
                  # the list of heads and bases of the set of interesting revisions.
                  # (head = revision in the set that has no descendant in the set;
                  #  base = revision in the set that has no ancestor in the set)
                  tostrip = set(striplist)
                  saveheads = set(saverevs)
                  for r in cl.revs(start=striprev + 1):
                      if any(p in tostrip for p in cl.parentrevs(r)):
                          tostrip.add(r)
                      if r not in tostrip:
                          saverevs.add(r)
                          saveheads.difference_update(cl.parentrevs(r))
                          saveheads.add(r)
                  saveheads = [cl.node(r) for r in saveheads]
                  # compute base nodes
                  if saverevs:
                      descendants = set(cl.descendants(saverevs))
                      saverevs.difference_update(descendants)
                  savebases = [cl.node(r) for r in saverevs]
                  stripbases = [cl.node(r) for r in tostrip]
                  stripobsidx = obsmarkers = ()
                  if repo.ui.configbool(b'devel', b'strip-obsmarkers'):
                      obsmarkers = obsutil.exclusivemarkers(repo, stripbases)
                  if obsmarkers:
                      stripobsidx = [
                          i for i, m in enumerate(repo.obsstore) if m in obsmarkers
                      ]
                  newbmtarget, updatebm = _bookmarkmovements(repo, tostrip)
                  backupfile = None
                  node = nodelist[-1]
                  if backup:
                      backupfile = _createstripbackup(repo, stripbases, node, topic)
                  # create a changegroup for all the branches we need to keep
                  tmpbundlefile = None
                  if saveheads:
                      # do not compress temporary bundle if we remove it from disk later
                      #
                      # We do not include obsolescence, it might re-introduce prune markers
                      # we are trying to strip.  This is harmless since the stripped markers
                      # are already backed up and we did not touched the markers for the
                      # saved changesets.
                      tmpbundlefile = backupbundle(
                          repo,
                          savebases,
                          saveheads,
                          node,
                          b'temp',
                          compress=False,
                          obsolescence=False,
                          tmp_backup=True,
                      )
                  with ui.uninterruptible():
                      try:
                          with repo.transaction(b"strip") as tr:
                              # TODO this code violates the interface abstraction of the
                              # transaction and makes assumptions that file storage is
                              # using append-only files. We'll need some kind of storage
                              # API to handle stripping for us.
                              oldfiles = set(tr._offsetmap.keys())
                              oldfiles.update(tr._newfiles)
                              tr.startgroup()
                              cl.strip(striprev, tr)
                              stripmanifest(repo, striprev, tr, files)
                              for fn in files:
                                  repo.file(fn).strip(striprev, tr)
                              tr.endgroup()
                              entries = tr.readjournal()
                              for file, troffset in entries:
                                  if file in oldfiles:
                                      continue
                                  with repo.svfs(file, b'a', checkambig=True) as fp:
                                      fp.truncate(troffset)
                                  if troffset == 0:
                                      repo.store.markremoved(file)
                              deleteobsmarkers(repo.obsstore, stripobsidx)
                              del repo.obsstore
                              repo.invalidatevolatilesets()
                              repo._phasecache.filterunknown(repo)
                          if tmpbundlefile:
                              ui.note(_(b"adding branch\n"))
                              f = vfs.open(tmpbundlefile, b"rb")
                              gen = exchange.readbundle(ui, f, tmpbundlefile, vfs)
                              # silence internal shuffling chatter
                              maybe_silent = (
                                  repo.ui.silent()
                                  if not repo.ui.verbose
                                  else util.nullcontextmanager()
                              )
                              with maybe_silent:
                                  tmpbundleurl = b'bundle:' + vfs.join(tmpbundlefile)
                                  txnname = b'strip'
                                  if not isinstance(gen, bundle2.unbundle20):
                                      txnname = b"strip\n%s" % urlutil.hidepassword(
                                          tmpbundleurl
                                      )
                                  with repo.transaction(txnname) as tr:
                                      bundle2.applybundle(
                                          repo, gen, tr, source=b'strip', url=tmpbundleurl
                                      )
                              f.close()
                          with repo.transaction(b'repair') as tr:
                              bmchanges = [(m, repo[newbmtarget].node()) for m in updatebm]
                              repo._bookmarks.applychanges(repo, tr, bmchanges)
                          transaction.cleanup_undo_files(repo.ui.warn, repo.vfs_map)
                      except:  # re-raises
                          if backupfile:
                              ui.warn(
                                  _(b"strip failed, backup bundle stored in '%s'\n")
                                  % vfs.join(backupfile)
                              )
                          if tmpbundlefile:
                              ui.warn(
                                  _(b"strip failed, unrecovered changes stored in '%s'\n")
                                  % vfs.join(tmpbundlefile)
                              )
                              ui.warn(
                                  _(
                                      b"(fix the problem, then recover the changesets with "
                                      b"\"hg unbundle '%s'\")\n"
                                  )
                                  % vfs.join(tmpbundlefile)
                              )
                          raise
                      else:
                          if tmpbundlefile:
                              # Remove temporary bundle only if there were no exceptions
                              vfs.unlink(tmpbundlefile)
                  repo.destroyed()
                  # return the backup file path (or None if 'backup' was False) so
                  # extensions can use it
                  return backupfile
              def softstrip(ui, repo, nodelist, backup=True, topic=b'backup'):
                  """perform a "soft" strip using the archived phase"""
                  tostrip = [c.node() for c in repo.set(b'sort(%ln::)', nodelist)]
                  if not tostrip:
                      return None
                  backupfile = None
                  if backup:
                      node = tostrip[0]
                      backupfile = _createstripbackup(repo, tostrip, node, topic)
                  newbmtarget, updatebm = _bookmarkmovements(repo, tostrip)
                  with repo.transaction(b'strip') as tr:
                      phases.retractboundary(repo, tr, phases.archived, tostrip)
                      bmchanges = [(m, repo[newbmtarget].node()) for m in updatebm]
                      repo._bookmarks.applychanges(repo, tr, bmchanges)
                  return backupfile
              def _bookmarkmovements(repo, tostrip):
                  # compute necessary bookmark movement
                  bm = repo._bookmarks
                  updatebm = []
                  for m in bm:
                      rev = repo[bm[m]].rev()
                      if rev in tostrip:
                          updatebm.append(m)
                  newbmtarget = None
                  # If we need to move bookmarks, compute bookmark
                  # targets. Otherwise we can skip doing this logic.
                  if updatebm:
                      # For a set s, max(parents(s) - s) is the same as max(heads(::s - s)),
                      # but is much faster
                      newbmtarget = repo.revs(b'max(parents(%ld) - (%ld))', tostrip, tostrip)
                      if newbmtarget:
                          newbmtarget = repo[newbmtarget.first()].node()
                      else:
                          newbmtarget = b'.'
                  return newbmtarget, updatebm
              def _createstripbackup(repo, stripbases, node, topic):
                  # backup the changeset we are about to strip
                  vfs = repo.vfs
                  unfi = repo.unfiltered()
                  to_node = unfi.changelog.node
                  # internal changeset are internal implementation details that should not
                  # leave the repository and not be exposed to the users. In addition feature
                  # using them requires to be resistant to strip. See test case for more
                  # details.
                  all_backup = unfi.revs(
                      b"(%ln)::(%ld) and not _internal()",
                      stripbases,
                      unfi.changelog.headrevs(),
                  )
                  if not all_backup:
                      return None
                  def to_nodes(revs):
                      return [to_node(r) for r in revs]
                  bases = to_nodes(unfi.revs("roots(%ld)", all_backup))
                  heads = to_nodes(unfi.revs("heads(%ld)", all_backup))
                  backupfile = backupbundle(repo, bases, heads, node, topic)
                  repo.ui.status(_(b"saved backup bundle to %s\n") % vfs.join(backupfile))
                  repo.ui.log(
                      b"backupbundle", b"saved backup bundle to %s\n", vfs.join(backupfile)
                  )
                  return backupfile
              def safestriproots(ui, repo, nodes):
                  """return list of roots of nodes where descendants are covered by nodes"""
                  torev = repo.unfiltered().changelog.rev
                  revs = {torev(n) for n in nodes}
                  # tostrip = wanted - unsafe = wanted - ancestors(orphaned)
                  # orphaned = affected - wanted
                  # affected = descendants(roots(wanted))
                  # wanted = revs
                  revset = b'%ld - ( ::( (roots(%ld):: and not _phase(%s)) -%ld) )'
                  tostrip = set(repo.revs(revset, revs, revs, phases.internal, revs))
                  notstrip = revs - tostrip
                  if notstrip:
                      nodestr = b', '.join(sorted(short(repo[n].node()) for n in notstrip))
                      ui.warn(
                          _(b'warning: orphaned descendants detected, not stripping %s\n')
                          % nodestr
                      )
                  return [c.node() for c in repo.set(b'roots(%ld)', tostrip)]
              class stripcallback:
                  """used as a transaction postclose callback"""
                  def __init__(self, ui, repo, backup, topic):
                      self.ui = ui
                      self.repo = repo
                      self.backup = backup
                      self.topic = topic or b'backup'
                      self.nodelist = []
                  def addnodes(self, nodes):
                      self.nodelist.extend(nodes)
                  def __call__(self, tr):
                      roots = safestriproots(self.ui, self.repo, self.nodelist)
                      if roots:
                          strip(self.ui, self.repo, roots, self.backup, self.topic)
              def delayedstrip(ui, repo, nodelist, topic=None, backup=True):
                  """like strip, but works inside transaction and won't strip irreverent revs
                  nodelist must explicitly contain all descendants. Otherwise a warning will
                  be printed that some nodes are not stripped.
                  Will do a backup if `backup` is True. The last non-None "topic" will be
                  used as the backup topic name. The default backup topic name is "backup".
                  """
                  tr = repo.currenttransaction()
                  if not tr:
                      nodes = safestriproots(ui, repo, nodelist)
                      return strip(ui, repo, nodes, backup=backup, topic=topic)
                  # transaction postclose callbacks are called in alphabet order.
                  # use '\xff' as prefix so we are likely to be called last.
                  callback = tr.getpostclose(b'\xffstrip')
                  if callback is None:
                      callback = stripcallback(ui, repo, backup=backup, topic=topic)
                      tr.addpostclose(b'\xffstrip', callback)
                  if topic:
                      callback.topic = topic
                  callback.addnodes(nodelist)
              def stripmanifest(repo, striprev, tr, files):
                  for revlog in manifestrevlogs(repo):
                      revlog.strip(striprev, tr)
              def manifestrevlogs(repo):
                  yield repo.manifestlog.getstorage(b'')
                  if scmutil.istreemanifest(repo):
                      # This logic is safe if treemanifest isn't enabled, but also
                      # pointless, so we skip it if treemanifest isn't enabled.
-                     for entry in repo.store.datafiles():
+                     for entry in repo.store.data_entries():
                          if not entry.is_revlog:
                              continue
                          if entry.revlog_type == store.FILEFLAGS_MANIFESTLOG:
                              yield repo.manifestlog.getstorage(entry.target_id)
              def rebuildfncache(ui, repo, only_data=False):
                  """Rebuilds the fncache file from repo history.
                  Missing entries will be added. Extra entries will be removed.
                  """
                  repo = repo.unfiltered()
                  if requirements.FNCACHE_REQUIREMENT not in repo.requirements:
                      ui.warn(
                          _(
                              b'(not rebuilding fncache because repository does not '
                              b'support fncache)\n'
                          )
                      )
                      return
                  with repo.lock():
                      fnc = repo.store.fncache
                      fnc.ensureloaded(warn=ui.warn)
                      oldentries = set(fnc.entries)
                      newentries = set()
                      seenfiles = set()
                      if only_data:
                          # Trust the listing of .i from the fncache, but not the .d. This is
                          # much faster, because we only need to stat every possible .d files,
                          # instead of reading the full changelog
                          for f in fnc:
                              if f[:5] == b'data/' and f[-2:] == b'.i':
                                  seenfiles.add(f[5:-2])
                                  newentries.add(f)
                                  dataf = f[:-2] + b'.d'
                                  if repo.store._exists(dataf):
                                      newentries.add(dataf)
                      else:
                          progress = ui.makeprogress(
                              _(b'rebuilding'), unit=_(b'changesets'), total=len(repo)
                          )
                          for rev in repo:
                              progress.update(rev)
                              ctx = repo[rev]
                              for f in ctx.files():
                                  # This is to minimize I/O.
                                  if f in seenfiles:
                                      continue
                                  seenfiles.add(f)
                                  i = b'data/%s.i' % f
                                  d = b'data/%s.d' % f
                                  if repo.store._exists(i):
                                      newentries.add(i)
                                  if repo.store._exists(d):
                                      newentries.add(d)
                          progress.complete()
                      if requirements.TREEMANIFEST_REQUIREMENT in repo.requirements:
                          # This logic is safe if treemanifest isn't enabled, but also
                          # pointless, so we skip it if treemanifest isn't enabled.
                          for dir in pathutil.dirs(seenfiles):
                              i = b'meta/%s/00manifest.i' % dir
                              d = b'meta/%s/00manifest.d' % dir
                              if repo.store._exists(i):
                                  newentries.add(i)
                              if repo.store._exists(d):
                                  newentries.add(d)
                      addcount = len(newentries - oldentries)
                      removecount = len(oldentries - newentries)
                      for p in sorted(oldentries - newentries):
                          ui.write(_(b'removing %s\n') % p)
                      for p in sorted(newentries - oldentries):
                          ui.write(_(b'adding %s\n') % p)
                      if addcount or removecount:
                          ui.write(
                              _(b'%d items added, %d removed from fncache\n')
                              % (addcount, removecount)
                          )
                          fnc.entries = newentries
                          fnc._dirty = True
                          with repo.transaction(b'fncache') as tr:
                              fnc.write(tr)
                      else:
                          ui.write(_(b'fncache already up to date\n'))
              def deleteobsmarkers(obsstore, indices):
                  """Delete some obsmarkers from obsstore and return how many were deleted
                  'indices' is a list of ints which are the indices
                  of the markers to be deleted.
                  Every invocation of this function completely rewrites the obsstore file,
                  skipping the markers we want to be removed. The new temporary file is
                  created, remaining markers are written there and on .close() this file
                  gets atomically renamed to obsstore, thus guaranteeing consistency."""
                  if not indices:
                      # we don't want to rewrite the obsstore with the same content
                      return
                  left = []
                  current = obsstore._all
                  n = 0
                  for i, m in enumerate(current):
                      if i in indices:
                          n += 1
                          continue
                      left.append(m)
                  newobsstorefile = obsstore.svfs(b'obsstore', b'w', atomictemp=True)
                  for bytes in obsolete.encodemarkers(left, True, obsstore._version):
                      newobsstorefile.write(bytes)
                  newobsstorefile.close()
                  return n

mercurial/revlogutils/rewrite.py

0 +1 -1

              # censor code related to censoring revision
              # coding: utf8
              #
              # Copyright 2021 Pierre-Yves David <pierre-yves.david@octobus.net>
              # Copyright 2015 Google, Inc <martinvonz@google.com>
              #
              # This software may be used and distributed according to the terms of the
              # GNU General Public License version 2 or any later version.
              import binascii
              import contextlib
              import os
              import struct
              from ..node import (
                  nullrev,
              )
              from .constants import (
                  COMP_MODE_PLAIN,
                  ENTRY_DATA_COMPRESSED_LENGTH,
                  ENTRY_DATA_COMPRESSION_MODE,
                  ENTRY_DATA_OFFSET,
                  ENTRY_DATA_UNCOMPRESSED_LENGTH,
                  ENTRY_DELTA_BASE,
                  ENTRY_LINK_REV,
                  ENTRY_NODE_ID,
                  ENTRY_PARENT_1,
                  ENTRY_PARENT_2,
                  ENTRY_SIDEDATA_COMPRESSED_LENGTH,
                  ENTRY_SIDEDATA_COMPRESSION_MODE,
                  ENTRY_SIDEDATA_OFFSET,
                  REVIDX_ISCENSORED,
                  REVLOGV0,
                  REVLOGV1,
              )
              from ..i18n import _
              from .. import (
                  error,
                  mdiff,
                  pycompat,
                  revlogutils,
                  util,
              )
              from ..utils import (
                  storageutil,
              )
              from . import (
                  constants,
                  deltas,
              )
              def v1_censor(rl, tr, censornode, tombstone=b''):
                  """censors a revision in a "version 1" revlog"""
                  assert rl._format_version == constants.REVLOGV1, rl._format_version
                  # avoid cycle
                  from .. import revlog
                  censorrev = rl.rev(censornode)
                  tombstone = storageutil.packmeta({b'censored': tombstone}, b'')
                  # Rewriting the revlog in place is hard. Our strategy for censoring is
                  # to create a new revlog, copy all revisions to it, then replace the
                  # revlogs on transaction close.
                  #
                  # This is a bit dangerous. We could easily have a mismatch of state.
                  newrl = revlog.revlog(
                      rl.opener,
                      target=rl.target,
                      radix=rl.radix,
                      postfix=b'tmpcensored',
                      censorable=True,
                  )
                  newrl._format_version = rl._format_version
                  newrl._format_flags = rl._format_flags
                  newrl._generaldelta = rl._generaldelta
                  newrl._parse_index = rl._parse_index
                  for rev in rl.revs():
                      node = rl.node(rev)
                      p1, p2 = rl.parents(node)
                      if rev == censorrev:
                          newrl.addrawrevision(
                              tombstone,
                              tr,
                              rl.linkrev(censorrev),
                              p1,
                              p2,
                              censornode,
                              constants.REVIDX_ISCENSORED,
                          )
                          if newrl.deltaparent(rev) != nullrev:
                              m = _(b'censored revision stored as delta; cannot censor')
                              h = _(
                                  b'censoring of revlogs is not fully implemented;'
                                  b' please report this bug'
                              )
                              raise error.Abort(m, hint=h)
                          continue
                      if rl.iscensored(rev):
                          if rl.deltaparent(rev) != nullrev:
                              m = _(
                                  b'cannot censor due to censored '
                                  b'revision having delta stored'
                              )
                              raise error.Abort(m)
                          rawtext = rl._chunk(rev)
                      else:
                          rawtext = rl.rawdata(rev)
                      newrl.addrawrevision(
                          rawtext, tr, rl.linkrev(rev), p1, p2, node, rl.flags(rev)
                      )
                  tr.addbackup(rl._indexfile, location=b'store')
                  if not rl._inline:
                      tr.addbackup(rl._datafile, location=b'store')
                  rl.opener.rename(newrl._indexfile, rl._indexfile)
                  if not rl._inline:
                      rl.opener.rename(newrl._datafile, rl._datafile)
                  rl.clearcaches()
                  rl._loadindex()
              def v2_censor(revlog, tr, censornode, tombstone=b''):
                  """censors a revision in a "version 2" revlog"""
                  assert revlog._format_version != REVLOGV0, revlog._format_version
                  assert revlog._format_version != REVLOGV1, revlog._format_version
                  censor_revs = {revlog.rev(censornode)}
                  _rewrite_v2(revlog, tr, censor_revs, tombstone)
              def _rewrite_v2(revlog, tr, censor_revs, tombstone=b''):
                  """rewrite a revlog to censor some of its content
                  General principle
                  We create new revlog files (index/data/sidedata) to copy the content of
                  the existing data without the censored data.
                  We need to recompute new delta for any revision that used the censored
                  revision as delta base. As the cumulative size of the new delta may be
                  large, we store them in a temporary file until they are stored in their
                  final destination.
                  All data before the censored data can be blindly copied. The rest needs
                  to be copied as we go and the associated index entry needs adjustement.
                  """
                  assert revlog._format_version != REVLOGV0, revlog._format_version
                  assert revlog._format_version != REVLOGV1, revlog._format_version
                  old_index = revlog.index
                  docket = revlog._docket
                  tombstone = storageutil.packmeta({b'censored': tombstone}, b'')
                  first_excl_rev = min(censor_revs)
                  first_excl_entry = revlog.index[first_excl_rev]
                  index_cutoff = revlog.index.entry_size * first_excl_rev
                  data_cutoff = first_excl_entry[ENTRY_DATA_OFFSET] >> 16
                  sidedata_cutoff = revlog.sidedata_cut_off(first_excl_rev)
                  with pycompat.unnamedtempfile(mode=b"w+b") as tmp_storage:
                      # rev → (new_base, data_start, data_end, compression_mode)
                      rewritten_entries = _precompute_rewritten_delta(
                          revlog,
                          old_index,
                          censor_revs,
                          tmp_storage,
                      )
                      all_files = _setup_new_files(
                          revlog,
                          index_cutoff,
                          data_cutoff,
                          sidedata_cutoff,
                      )
                      # we dont need to open the old index file since its content already
                      # exist in a usable form in `old_index`.
                      with all_files() as open_files:
                          (
                              old_data_file,
                              old_sidedata_file,
                              new_index_file,
                              new_data_file,
                              new_sidedata_file,
                          ) = open_files
                          # writing the censored revision
                          # Writing all subsequent revisions
                          for rev in range(first_excl_rev, len(old_index)):
                              if rev in censor_revs:
                                  _rewrite_censor(
                                      revlog,
                                      old_index,
                                      open_files,
                                      rev,
                                      tombstone,
                                  )
                              else:
                                  _rewrite_simple(
                                      revlog,
                                      old_index,
                                      open_files,
                                      rev,
                                      rewritten_entries,
                                      tmp_storage,
                                  )
                  docket.write(transaction=None, stripping=True)
              def _precompute_rewritten_delta(
                  revlog,
                  old_index,
                  excluded_revs,
                  tmp_storage,
              ):
                  """Compute new delta for revisions whose delta is based on revision that
                  will not survive as is.
                  Return a mapping: {rev → (new_base, data_start, data_end, compression_mode)}
                  """
                  dc = deltas.deltacomputer(revlog)
                  rewritten_entries = {}
                  first_excl_rev = min(excluded_revs)
                  with revlog._segmentfile._open_read() as dfh:
                      for rev in range(first_excl_rev, len(old_index)):
                          if rev in excluded_revs:
                              # this revision will be preserved as is, so we don't need to
                              # consider recomputing a delta.
                              continue
                          entry = old_index[rev]
                          if entry[ENTRY_DELTA_BASE] not in excluded_revs:
                              continue
                          # This is a revision that use the censored revision as the base
                          # for its delta. We need a need new deltas
                          if entry[ENTRY_DATA_UNCOMPRESSED_LENGTH] == 0:
                              # this revision is empty, we can delta against nullrev
                              rewritten_entries[rev] = (nullrev, 0, 0, COMP_MODE_PLAIN)
                          else:
                              text = revlog.rawdata(rev, _df=dfh)
                              info = revlogutils.revisioninfo(
                                  node=entry[ENTRY_NODE_ID],
                                  p1=revlog.node(entry[ENTRY_PARENT_1]),
                                  p2=revlog.node(entry[ENTRY_PARENT_2]),
                                  btext=[text],
                                  textlen=len(text),
                                  cachedelta=None,
                                  flags=entry[ENTRY_DATA_OFFSET] & 0xFFFF,
                              )
                              d = dc.finddeltainfo(
                                  info, dfh, excluded_bases=excluded_revs, target_rev=rev
                              )
                              default_comp = revlog._docket.default_compression_header
                              comp_mode, d = deltas.delta_compression(default_comp, d)
                              # using `tell` is a bit lazy, but we are not here for speed
                              start = tmp_storage.tell()
                              tmp_storage.write(d.data[1])
                              end = tmp_storage.tell()
                              rewritten_entries[rev] = (d.base, start, end, comp_mode)
                  return rewritten_entries
              def _setup_new_files(
                  revlog,
                  index_cutoff,
                  data_cutoff,
                  sidedata_cutoff,
              ):
                  """
                  return a context manager to open all the relevant files:
                  - old_data_file,
                  - old_sidedata_file,
                  - new_index_file,
                  - new_data_file,
                  - new_sidedata_file,
                  The old_index_file is not here because it is accessed through the
                  `old_index` object if the caller function.
                  """
                  docket = revlog._docket
                  old_index_filepath = revlog.opener.join(docket.index_filepath())
                  old_data_filepath = revlog.opener.join(docket.data_filepath())
                  old_sidedata_filepath = revlog.opener.join(docket.sidedata_filepath())
                  new_index_filepath = revlog.opener.join(docket.new_index_file())
                  new_data_filepath = revlog.opener.join(docket.new_data_file())
                  new_sidedata_filepath = revlog.opener.join(docket.new_sidedata_file())
                  util.copyfile(old_index_filepath, new_index_filepath, nb_bytes=index_cutoff)
                  util.copyfile(old_data_filepath, new_data_filepath, nb_bytes=data_cutoff)
                  util.copyfile(
                      old_sidedata_filepath,
                      new_sidedata_filepath,
                      nb_bytes=sidedata_cutoff,
                  )
                  revlog.opener.register_file(docket.index_filepath())
                  revlog.opener.register_file(docket.data_filepath())
                  revlog.opener.register_file(docket.sidedata_filepath())
                  docket.index_end = index_cutoff
                  docket.data_end = data_cutoff
                  docket.sidedata_end = sidedata_cutoff
                  # reload the revlog internal information
                  revlog.clearcaches()
                  revlog._loadindex(docket=docket)
                  @contextlib.contextmanager
                  def all_files_opener():
                      # hide opening in an helper function to please check-code, black
                      # and various python version at the same time
                      with open(old_data_filepath, 'rb') as old_data_file:
                          with open(old_sidedata_filepath, 'rb') as old_sidedata_file:
                              with open(new_index_filepath, 'r+b') as new_index_file:
                                  with open(new_data_filepath, 'r+b') as new_data_file:
                                      with open(
                                          new_sidedata_filepath, 'r+b'
                                      ) as new_sidedata_file:
                                          new_index_file.seek(0, os.SEEK_END)
                                          assert new_index_file.tell() == index_cutoff
                                          new_data_file.seek(0, os.SEEK_END)
                                          assert new_data_file.tell() == data_cutoff
                                          new_sidedata_file.seek(0, os.SEEK_END)
                                          assert new_sidedata_file.tell() == sidedata_cutoff
                                          yield (
                                              old_data_file,
                                              old_sidedata_file,
                                              new_index_file,
                                              new_data_file,
                                              new_sidedata_file,
                                          )
                  return all_files_opener
              def _rewrite_simple(
                  revlog,
                  old_index,
                  all_files,
                  rev,
                  rewritten_entries,
                  tmp_storage,
              ):
                  """append a normal revision to the index after the rewritten one(s)"""
                  (
                      old_data_file,
                      old_sidedata_file,
                      new_index_file,
                      new_data_file,
                      new_sidedata_file,
                  ) = all_files
                  entry = old_index[rev]
                  flags = entry[ENTRY_DATA_OFFSET] & 0xFFFF
                  old_data_offset = entry[ENTRY_DATA_OFFSET] >> 16
                  if rev not in rewritten_entries:
                      old_data_file.seek(old_data_offset)
                      new_data_size = entry[ENTRY_DATA_COMPRESSED_LENGTH]
                      new_data = old_data_file.read(new_data_size)
                      data_delta_base = entry[ENTRY_DELTA_BASE]
                      d_comp_mode = entry[ENTRY_DATA_COMPRESSION_MODE]
                  else:
                      (
                          data_delta_base,
                          start,
                          end,
                          d_comp_mode,
                      ) = rewritten_entries[rev]
                      new_data_size = end - start
                      tmp_storage.seek(start)
                      new_data = tmp_storage.read(new_data_size)
                  # It might be faster to group continuous read/write operation,
                  # however, this is censor, an operation that is not focussed
                  # around stellar performance. So I have not written this
                  # optimisation yet.
                  new_data_offset = new_data_file.tell()
                  new_data_file.write(new_data)
                  sidedata_size = entry[ENTRY_SIDEDATA_COMPRESSED_LENGTH]
                  new_sidedata_offset = new_sidedata_file.tell()
                  if 0 < sidedata_size:
                      old_sidedata_offset = entry[ENTRY_SIDEDATA_OFFSET]
                      old_sidedata_file.seek(old_sidedata_offset)
                      new_sidedata = old_sidedata_file.read(sidedata_size)
                      new_sidedata_file.write(new_sidedata)
                  data_uncompressed_length = entry[ENTRY_DATA_UNCOMPRESSED_LENGTH]
                  sd_com_mode = entry[ENTRY_SIDEDATA_COMPRESSION_MODE]
                  assert data_delta_base <= rev, (data_delta_base, rev)
                  new_entry = revlogutils.entry(
                      flags=flags,
                      data_offset=new_data_offset,
                      data_compressed_length=new_data_size,
                      data_uncompressed_length=data_uncompressed_length,
                      data_delta_base=data_delta_base,
                      link_rev=entry[ENTRY_LINK_REV],
                      parent_rev_1=entry[ENTRY_PARENT_1],
                      parent_rev_2=entry[ENTRY_PARENT_2],
                      node_id=entry[ENTRY_NODE_ID],
                      sidedata_offset=new_sidedata_offset,
                      sidedata_compressed_length=sidedata_size,
                      data_compression_mode=d_comp_mode,
                      sidedata_compression_mode=sd_com_mode,
                  )
                  revlog.index.append(new_entry)
                  entry_bin = revlog.index.entry_binary(rev)
                  new_index_file.write(entry_bin)
                  revlog._docket.index_end = new_index_file.tell()
                  revlog._docket.data_end = new_data_file.tell()
                  revlog._docket.sidedata_end = new_sidedata_file.tell()
              def _rewrite_censor(
                  revlog,
                  old_index,
                  all_files,
                  rev,
                  tombstone,
              ):
                  """rewrite and append a censored revision"""
                  (
                      old_data_file,
                      old_sidedata_file,
                      new_index_file,
                      new_data_file,
                      new_sidedata_file,
                  ) = all_files
                  entry = old_index[rev]
                  # XXX consider trying the default compression too
                  new_data_size = len(tombstone)
                  new_data_offset = new_data_file.tell()
                  new_data_file.write(tombstone)
                  # we are not adding any sidedata as they might leak info about the censored version
                  link_rev = entry[ENTRY_LINK_REV]
                  p1 = entry[ENTRY_PARENT_1]
                  p2 = entry[ENTRY_PARENT_2]
                  new_entry = revlogutils.entry(
                      flags=constants.REVIDX_ISCENSORED,
                      data_offset=new_data_offset,
                      data_compressed_length=new_data_size,
                      data_uncompressed_length=new_data_size,
                      data_delta_base=rev,
                      link_rev=link_rev,
                      parent_rev_1=p1,
                      parent_rev_2=p2,
                      node_id=entry[ENTRY_NODE_ID],
                      sidedata_offset=0,
                      sidedata_compressed_length=0,
                      data_compression_mode=COMP_MODE_PLAIN,
                      sidedata_compression_mode=COMP_MODE_PLAIN,
                  )
                  revlog.index.append(new_entry)
                  entry_bin = revlog.index.entry_binary(rev)
                  new_index_file.write(entry_bin)
                  revlog._docket.index_end = new_index_file.tell()
                  revlog._docket.data_end = new_data_file.tell()
              def _get_filename_from_filelog_index(path):
                  # Drop the extension and the `data/` prefix
                  path_part = path.rsplit(b'.', 1)[0].split(b'/', 1)
                  if len(path_part) < 2:
                      msg = _(b"cannot recognize filelog from filename: '%s'")
                      msg %= path
                      raise error.Abort(msg)
                  return path_part[1]
              def _filelog_from_filename(repo, path):
                  """Returns the filelog for the given `path`. Stolen from `engine.py`"""
                  from .. import filelog  # avoid cycle
                  fl = filelog.filelog(repo.svfs, path)
                  return fl
              def _write_swapped_parents(repo, rl, rev, offset, fp):
                  """Swaps p1 and p2 and overwrites the revlog entry for `rev` in `fp`"""
                  from ..pure import parsers  # avoid cycle
                  if repo._currentlock(repo._lockref) is None:
                      # Let's be paranoid about it
                      msg = "repo needs to be locked to rewrite parents"
                      raise error.ProgrammingError(msg)
                  index_format = parsers.IndexObject.index_format
                  entry = rl.index[rev]
                  new_entry = list(entry)
                  new_entry[5], new_entry[6] = entry[6], entry[5]
                  packed = index_format.pack(*new_entry[:8])
                  fp.seek(offset)
                  fp.write(packed)
              def _reorder_filelog_parents(repo, fl, to_fix):
                  """
                  Swaps p1 and p2 for all `to_fix` revisions of filelog `fl` and writes the
                  new version to disk, overwriting the old one with a rename.
                  """
                  from ..pure import parsers  # avoid cycle
                  ui = repo.ui
                  assert len(to_fix) > 0
                  rl = fl._revlog
                  if rl._format_version != constants.REVLOGV1:
                      msg = "expected version 1 revlog, got version '%d'" % rl._format_version
                      raise error.ProgrammingError(msg)
                  index_file = rl._indexfile
                  new_file_path = index_file + b'.tmp-parents-fix'
                  repaired_msg = _(b"repaired revision %d of 'filelog %s'\n")
                  with ui.uninterruptible():
                      try:
                          util.copyfile(
                              rl.opener.join(index_file),
                              rl.opener.join(new_file_path),
                              checkambig=rl._checkambig,
                          )
                          with rl.opener(new_file_path, mode=b"r+") as fp:
                              if rl._inline:
                                  index = parsers.InlinedIndexObject(fp.read())
                                  for rev in fl.revs():
                                      if rev in to_fix:
                                          offset = index._calculate_index(rev)
                                          _write_swapped_parents(repo, rl, rev, offset, fp)
                                          ui.write(repaired_msg % (rev, index_file))
                              else:
                                  index_format = parsers.IndexObject.index_format
                                  for rev in to_fix:
                                      offset = rev * index_format.size
                                      _write_swapped_parents(repo, rl, rev, offset, fp)
                                      ui.write(repaired_msg % (rev, index_file))
                          rl.opener.rename(new_file_path, index_file)
                          rl.clearcaches()
                          rl._loadindex()
                      finally:
                          util.tryunlink(new_file_path)
              def _is_revision_affected(fl, filerev, metadata_cache=None):
                  full_text = lambda: fl._revlog.rawdata(filerev)
                  parent_revs = lambda: fl._revlog.parentrevs(filerev)
                  return _is_revision_affected_inner(
                      full_text, parent_revs, filerev, metadata_cache
                  )
              def _is_revision_affected_inner(
                  full_text,
                  parents_revs,
                  filerev,
                  metadata_cache=None,
              ):
                  """Mercurial currently (5.9rc0) uses `p1 == nullrev and p2 != nullrev` as a
                  special meaning compared to the reverse in the context of filelog-based
                  copytracing. issue6528 exists because new code assumed that parent ordering
                  didn't matter, so this detects if the revision contains metadata (since
                  it's only used for filelog-based copytracing) and its parents are in the
                  "wrong" order."""
                  try:
                      raw_text = full_text()
                  except error.CensoredNodeError:
                      # We don't care about censored nodes as they never carry metadata
                      return False
                  # raw text can be a `memoryview`, which doesn't implement `startswith`
                  has_meta = bytes(raw_text[:2]) == b'\x01\n'
                  if metadata_cache is not None:
                      metadata_cache[filerev] = has_meta
                  if has_meta:
                      (p1, p2) = parents_revs()
                      if p1 != nullrev and p2 == nullrev:
                          return True
                  return False
              def _is_revision_affected_fast(repo, fl, filerev, metadata_cache):
                  rl = fl._revlog
                  is_censored = lambda: rl.iscensored(filerev)
                  delta_base = lambda: rl.deltaparent(filerev)
                  delta = lambda: rl._chunk(filerev)
                  full_text = lambda: rl.rawdata(filerev)
                  parent_revs = lambda: rl.parentrevs(filerev)
                  return _is_revision_affected_fast_inner(
                      is_censored,
                      delta_base,
                      delta,
                      full_text,
                      parent_revs,
                      filerev,
                      metadata_cache,
                  )
              def _is_revision_affected_fast_inner(
                  is_censored,
                  delta_base,
                  delta,
                  full_text,
                  parent_revs,
                  filerev,
                  metadata_cache,
              ):
                  """Optimization fast-path for `_is_revision_affected`.
                  `metadata_cache` is a dict of `{rev: has_metadata}` which allows any
                  revision to check if its base has metadata, saving computation of the full
                  text, instead looking at the current delta.
                  This optimization only works if the revisions are looked at in order."""
                  if is_censored():
                      # Censored revisions don't contain metadata, so they cannot be affected
                      metadata_cache[filerev] = False
                      return False
                  p1, p2 = parent_revs()
                  if p1 == nullrev or p2 != nullrev:
                      return False
                  delta_parent = delta_base()
                  parent_has_metadata = metadata_cache.get(delta_parent)
                  if parent_has_metadata is None:
                      return _is_revision_affected_inner(
                          full_text,
                          parent_revs,
                          filerev,
                          metadata_cache,
                      )
                  chunk = delta()
                  if not len(chunk):
                      # No diff for this revision
                      return parent_has_metadata
                  header_length = 12
                  if len(chunk) < header_length:
                      raise error.Abort(_(b"patch cannot be decoded"))
                  start, _end, _length = struct.unpack(b">lll", chunk[:header_length])
                  if start < 2:  # len(b'\x01\n') == 2
                      # This delta does *something* to the metadata marker (if any).
                      # Check it the slow way
                      is_affected = _is_revision_affected_inner(
                          full_text,
                          parent_revs,
                          filerev,
                          metadata_cache,
                      )
                      return is_affected
                  # The diff did not remove or add the metadata header, it's then in the same
                  # situation as its parent
                  metadata_cache[filerev] = parent_has_metadata
                  return parent_has_metadata
              def _from_report(ui, repo, context, from_report, dry_run):
                  """
                  Fix the revisions given in the `from_report` file, but still checks if the
                  revisions are indeed affected to prevent an unfortunate cyclic situation
                  where we'd swap well-ordered parents again.
                  See the doc for `debug_fix_issue6528` for the format documentation.
                  """
                  ui.write(_(b"loading report file '%s'\n") % from_report)
                  with context(), open(from_report, mode='rb') as f:
                      for line in f.read().split(b'\n'):
                          if not line:
                              continue
                          filenodes, filename = line.split(b' ', 1)
                          fl = _filelog_from_filename(repo, filename)
                          to_fix = set(
                              fl.rev(binascii.unhexlify(n)) for n in filenodes.split(b',')
                          )
                          excluded = set()
                          for filerev in to_fix:
                              if _is_revision_affected(fl, filerev):
                                  msg = b"found affected revision %d for filelog '%s'\n"
                                  ui.warn(msg % (filerev, filename))
                              else:
                                  msg = _(b"revision %s of file '%s' is not affected\n")
                                  msg %= (binascii.hexlify(fl.node(filerev)), filename)
                                  ui.warn(msg)
                                  excluded.add(filerev)
                          to_fix = to_fix - excluded
                          if not to_fix:
                              msg = _(b"no affected revisions were found for '%s'\n")
                              ui.write(msg % filename)
                              continue
                          if not dry_run:
                              _reorder_filelog_parents(repo, fl, sorted(to_fix))
              def filter_delta_issue6528(revlog, deltas_iter):
                  """filter incomind deltas to repaire issue 6528 on the fly"""
                  metadata_cache = {}
                  deltacomputer = deltas.deltacomputer(revlog)
                  for rev, d in enumerate(deltas_iter, len(revlog)):
                      (
                          node,
                          p1_node,
                          p2_node,
                          linknode,
                          deltabase,
                          delta,
                          flags,
                          sidedata,
                      ) = d
                      if not revlog.index.has_node(deltabase):
                          raise error.LookupError(
                              deltabase, revlog.radix, _(b'unknown parent')
                          )
                      base_rev = revlog.rev(deltabase)
                      if not revlog.index.has_node(p1_node):
                          raise error.LookupError(p1_node, revlog.radix, _(b'unknown parent'))
                      p1_rev = revlog.rev(p1_node)
                      if not revlog.index.has_node(p2_node):
                          raise error.LookupError(p2_node, revlog.radix, _(b'unknown parent'))
                      p2_rev = revlog.rev(p2_node)
                      is_censored = lambda: bool(flags & REVIDX_ISCENSORED)
                      delta_base = lambda: revlog.rev(delta_base)
                      delta_base = lambda: base_rev
                      parent_revs = lambda: (p1_rev, p2_rev)
                      def full_text():
                          # note: being able to reuse the full text computation in the
                          # underlying addrevision would be useful however this is a bit too
                          # intrusive the for the "quick" issue6528 we are writing before the
                          # 5.8 release
                          textlen = mdiff.patchedsize(revlog.size(base_rev), delta)
                          revinfo = revlogutils.revisioninfo(
                              node,
                              p1_node,
                              p2_node,
                              [None],
                              textlen,
                              (base_rev, delta),
                              flags,
                          )
                          # cached by the global "writing" context
                          assert revlog._writinghandles is not None
                          if revlog._inline:
                              fh = revlog._writinghandles[0]
                          else:
                              fh = revlog._writinghandles[1]
                          return deltacomputer.buildtext(revinfo, fh)
                      is_affected = _is_revision_affected_fast_inner(
                          is_censored,
                          delta_base,
                          lambda: delta,
                          full_text,
                          parent_revs,
                          rev,
                          metadata_cache,
                      )
                      if is_affected:
                          d = (
                              node,
                              p2_node,
                              p1_node,
                              linknode,
                              deltabase,
                              delta,
                              flags,
                              sidedata,
                          )
                      yield d
              def repair_issue6528(
                  ui, repo, dry_run=False, to_report=None, from_report=None, paranoid=False
              ):
                  @contextlib.contextmanager
                  def context():
                      if dry_run or to_report:  # No need for locking
                          yield
                      else:
                          with repo.wlock(), repo.lock():
                              yield
                  if from_report:
                      return _from_report(ui, repo, context, from_report, dry_run)
                  report_entries = []
                  with context():
                      files = list(
                          entry
-                         for entry in repo.store.datafiles()
+                         for entry in repo.store.data_entries()
                          if entry.is_revlog and entry.is_filelog
                      )
                      progress = ui.makeprogress(
                          _(b"looking for affected revisions"),
                          unit=_(b"filelogs"),
                          total=len(files),
                      )
                      found_nothing = True
                      for entry in files:
                          progress.increment()
                          filename = entry.target_id
                          fl = _filelog_from_filename(repo, entry.target_id)
                          # Set of filerevs (or hex filenodes if `to_report`) that need fixing
                          to_fix = set()
                          metadata_cache = {}
                          for filerev in fl.revs():
                              affected = _is_revision_affected_fast(
                                  repo, fl, filerev, metadata_cache
                              )
                              if paranoid:
                                  slow = _is_revision_affected(fl, filerev)
                                  if slow != affected:
                                      msg = _(b"paranoid check failed for '%s' at node %s")
                                      node = binascii.hexlify(fl.node(filerev))
                                      raise error.Abort(msg % (filename, node))
                              if affected:
                                  msg = b"found affected revision %d for file '%s'\n"
                                  ui.warn(msg % (filerev, filename))
                                  found_nothing = False
                                  if not dry_run:
                                      if to_report:
                                          to_fix.add(binascii.hexlify(fl.node(filerev)))
                                      else:
                                          to_fix.add(filerev)
                          if to_fix:
                              to_fix = sorted(to_fix)
                              if to_report:
                                  report_entries.append((filename, to_fix))
                              else:
                                  _reorder_filelog_parents(repo, fl, to_fix)
                      if found_nothing:
                          ui.write(_(b"no affected revisions were found\n"))
                      if to_report and report_entries:
                          with open(to_report, mode="wb") as f:
                              for path, to_fix in report_entries:
                                  f.write(b"%s %s\n" % (b",".join(to_fix), path))
                      progress.complete()

mercurial/store.py

0 +7 -5

              # store.py - repository store handling for Mercurial
              #
              # Copyright 2008 Olivia Mackall <olivia@selenic.com>
              #
              # This software may be used and distributed according to the terms of the
              # GNU General Public License version 2 or any later version.
              import collections
              import functools
              import os
              import re
              import stat
              from typing import Generator
              from .i18n import _
              from .pycompat import getattr
              from .thirdparty import attr
              from .node import hex
              from . import (
                  changelog,
                  error,
                  manifest,
                  policy,
                  pycompat,
                  util,
                  vfs as vfsmod,
              )
              from .utils import hashutil
              parsers = policy.importmod('parsers')
              # how much bytes should be read from fncache in one read
              # It is done to prevent loading large fncache files into memory
              fncache_chunksize = 10 ** 6
              def _match_tracked_entry(entry, matcher):
                  """parses a fncache entry and returns whether the entry is tracking a path
                  matched by matcher or not.
                  If matcher is None, returns True"""
                  if matcher is None:
                      return True
                  if entry.is_filelog:
                      return matcher(entry.target_id)
                  elif entry.is_manifestlog:
                      return matcher.visitdir(entry.target_id.rstrip(b'/'))
                  raise error.ProgrammingError(b"cannot process entry %r" % entry)
              # This avoids a collision between a file named foo and a dir named
              # foo.i or foo.d
              def _encodedir(path):
                  """
                  >>> _encodedir(b'data/foo.i')
                  'data/foo.i'
                  >>> _encodedir(b'data/foo.i/bla.i')
                  'data/foo.i.hg/bla.i'
                  >>> _encodedir(b'data/foo.i.hg/bla.i')
                  'data/foo.i.hg.hg/bla.i'
                  >>> _encodedir(b'data/foo.i\\ndata/foo.i/bla.i\\ndata/foo.i.hg/bla.i\\n')
                  'data/foo.i\\ndata/foo.i.hg/bla.i\\ndata/foo.i.hg.hg/bla.i\\n'
                  """
                  return (
                      path.replace(b".hg/", b".hg.hg/")
                      .replace(b".i/", b".i.hg/")
                      .replace(b".d/", b".d.hg/")
                  )
              encodedir = getattr(parsers, 'encodedir', _encodedir)
              def decodedir(path):
                  """
                  >>> decodedir(b'data/foo.i')
                  'data/foo.i'
                  >>> decodedir(b'data/foo.i.hg/bla.i')
                  'data/foo.i/bla.i'
                  >>> decodedir(b'data/foo.i.hg.hg/bla.i')
                  'data/foo.i.hg/bla.i'
                  """
                  if b".hg/" not in path:
                      return path
                  return (
                      path.replace(b".d.hg/", b".d/")
                      .replace(b".i.hg/", b".i/")
                      .replace(b".hg.hg/", b".hg/")
                  )
              def _reserved():
                  """characters that are problematic for filesystems
                  * ascii escapes (0..31)
                  * ascii hi (126..255)
                  * windows specials
                  these characters will be escaped by encodefunctions
                  """
                  winreserved = [ord(x) for x in u'\\:*?"<>|']
                  for x in range(32):
                      yield x
                  for x in range(126, 256):
                      yield x
                  for x in winreserved:
                      yield x
              def _buildencodefun():
                  """
                  >>> enc, dec = _buildencodefun()
                  >>> enc(b'nothing/special.txt')
                  'nothing/special.txt'
                  >>> dec(b'nothing/special.txt')
                  'nothing/special.txt'
                  >>> enc(b'HELLO')
                  '_h_e_l_l_o'
                  >>> dec(b'_h_e_l_l_o')
                  'HELLO'
                  >>> enc(b'hello:world?')
                  'hello~3aworld~3f'
                  >>> dec(b'hello~3aworld~3f')
                  'hello:world?'
                  >>> enc(b'the\\x07quick\\xADshot')
                  'the~07quick~adshot'
                  >>> dec(b'the~07quick~adshot')
                  'the\\x07quick\\xadshot'
                  """
                  e = b'_'
                  xchr = pycompat.bytechr
                  asciistr = list(map(xchr, range(127)))
                  capitals = list(range(ord(b"A"), ord(b"Z") + 1))
                  cmap = {x: x for x in asciistr}
                  for x in _reserved():
                      cmap[xchr(x)] = b"~%02x" % x
                  for x in capitals + [ord(e)]:
                      cmap[xchr(x)] = e + xchr(x).lower()
                  dmap = {}
                  for k, v in cmap.items():
                      dmap[v] = k
                  def decode(s):
                      i = 0
                      while i < len(s):
                          for l in range(1, 4):
                              try:
                                  yield dmap[s[i : i + l]]
                                  i += l
                                  break
                              except KeyError:
                                  pass
                          else:
                              raise KeyError
                  return (
                      lambda s: b''.join([cmap[s[c : c + 1]] for c in range(len(s))]),
                      lambda s: b''.join(list(decode(s))),
                  )
              _encodefname, _decodefname = _buildencodefun()
              def encodefilename(s):
                  """
                  >>> encodefilename(b'foo.i/bar.d/bla.hg/hi:world?/HELLO')
                  'foo.i.hg/bar.d.hg/bla.hg.hg/hi~3aworld~3f/_h_e_l_l_o'
                  """
                  return _encodefname(encodedir(s))
              def decodefilename(s):
                  """
                  >>> decodefilename(b'foo.i.hg/bar.d.hg/bla.hg.hg/hi~3aworld~3f/_h_e_l_l_o')
                  'foo.i/bar.d/bla.hg/hi:world?/HELLO'
                  """
                  return decodedir(_decodefname(s))
              def _buildlowerencodefun():
                  """
                  >>> f = _buildlowerencodefun()
                  >>> f(b'nothing/special.txt')
                  'nothing/special.txt'
                  >>> f(b'HELLO')
                  'hello'
                  >>> f(b'hello:world?')
                  'hello~3aworld~3f'
                  >>> f(b'the\\x07quick\\xADshot')
                  'the~07quick~adshot'
                  """
                  xchr = pycompat.bytechr
                  cmap = {xchr(x): xchr(x) for x in range(127)}
                  for x in _reserved():
                      cmap[xchr(x)] = b"~%02x" % x
                  for x in range(ord(b"A"), ord(b"Z") + 1):
                      cmap[xchr(x)] = xchr(x).lower()
                  def lowerencode(s):
                      return b"".join([cmap[c] for c in pycompat.iterbytestr(s)])
                  return lowerencode
              lowerencode = getattr(parsers, 'lowerencode', None) or _buildlowerencodefun()
              # Windows reserved names: con, prn, aux, nul, com1..com9, lpt1..lpt9
              _winres3 = (b'aux', b'con', b'prn', b'nul')  # length 3
              _winres4 = (b'com', b'lpt')  # length 4 (with trailing 1..9)
              def _auxencode(path, dotencode):
                  """
                  Encodes filenames containing names reserved by Windows or which end in
                  period or space. Does not touch other single reserved characters c.
                  Specifically, c in '\\:*?"<>|' or ord(c) <= 31 are *not* encoded here.
                  Additionally encodes space or period at the beginning, if dotencode is
                  True. Parameter path is assumed to be all lowercase.
                  A segment only needs encoding if a reserved name appears as a
                  basename (e.g. "aux", "aux.foo"). A directory or file named "foo.aux"
                  doesn't need encoding.
                  >>> s = b'.foo/aux.txt/txt.aux/con/prn/nul/foo.'
                  >>> _auxencode(s.split(b'/'), True)
                  ['~2efoo', 'au~78.txt', 'txt.aux', 'co~6e', 'pr~6e', 'nu~6c', 'foo~2e']
                  >>> s = b'.com1com2/lpt9.lpt4.lpt1/conprn/com0/lpt0/foo.'
                  >>> _auxencode(s.split(b'/'), False)
                  ['.com1com2', 'lp~749.lpt4.lpt1', 'conprn', 'com0', 'lpt0', 'foo~2e']
                  >>> _auxencode([b'foo. '], True)
                  ['foo.~20']
                  >>> _auxencode([b' .foo'], True)
                  ['~20.foo']
                  """
                  for i, n in enumerate(path):
                      if not n:
                          continue
                      if dotencode and n[0] in b'. ':
                          n = b"~%02x" % ord(n[0:1]) + n[1:]
                          path[i] = n
                      else:
                          l = n.find(b'.')
                          if l == -1:
                              l = len(n)
                          if (l == 3 and n[:3] in _winres3) or (
                              l == 4
                              and n[3:4] <= b'9'
                              and n[3:4] >= b'1'
                              and n[:3] in _winres4
                          ):
                              # encode third letter ('aux' -> 'au~78')
                              ec = b"~%02x" % ord(n[2:3])
                              n = n[0:2] + ec + n[3:]
                              path[i] = n
                      if n[-1] in b'. ':
                          # encode last period or space ('foo...' -> 'foo..~2e')
                          path[i] = n[:-1] + b"~%02x" % ord(n[-1:])
                  return path
              _maxstorepathlen = 120
              _dirprefixlen = 8
              _maxshortdirslen = 8 * (_dirprefixlen + 1) - 4
              def _hashencode(path, dotencode):
                  digest = hex(hashutil.sha1(path).digest())
                  le = lowerencode(path[5:]).split(b'/')  # skips prefix 'data/' or 'meta/'
                  parts = _auxencode(le, dotencode)
                  basename = parts[-1]
                  _root, ext = os.path.splitext(basename)
                  sdirs = []
                  sdirslen = 0
                  for p in parts[:-1]:
                      d = p[:_dirprefixlen]
                      if d[-1] in b'. ':
                          # Windows can't access dirs ending in period or space
                          d = d[:-1] + b'_'
                      if sdirslen == 0:
                          t = len(d)
                      else:
                          t = sdirslen + 1 + len(d)
                          if t > _maxshortdirslen:
                              break
                      sdirs.append(d)
                      sdirslen = t
                  dirs = b'/'.join(sdirs)
                  if len(dirs) > 0:
                      dirs += b'/'
                  res = b'dh/' + dirs + digest + ext
                  spaceleft = _maxstorepathlen - len(res)
                  if spaceleft > 0:
                      filler = basename[:spaceleft]
                      res = b'dh/' + dirs + filler + digest + ext
                  return res
              def _hybridencode(path, dotencode):
                  """encodes path with a length limit
                  Encodes all paths that begin with 'data/', according to the following.
                  Default encoding (reversible):
                  Encodes all uppercase letters 'X' as '_x'. All reserved or illegal
                  characters are encoded as '~xx', where xx is the two digit hex code
                  of the character (see encodefilename).
                  Relevant path components consisting of Windows reserved filenames are
                  masked by encoding the third character ('aux' -> 'au~78', see _auxencode).
                  Hashed encoding (not reversible):
                  If the default-encoded path is longer than _maxstorepathlen, a
                  non-reversible hybrid hashing of the path is done instead.
                  This encoding uses up to _dirprefixlen characters of all directory
                  levels of the lowerencoded path, but not more levels than can fit into
                  _maxshortdirslen.
                  Then follows the filler followed by the sha digest of the full path.
                  The filler is the beginning of the basename of the lowerencoded path
                  (the basename is everything after the last path separator). The filler
                  is as long as possible, filling in characters from the basename until
                  the encoded path has _maxstorepathlen characters (or all chars of the
                  basename have been taken).
                  The extension (e.g. '.i' or '.d') is preserved.
                  The string 'data/' at the beginning is replaced with 'dh/', if the hashed
                  encoding was used.
                  """
                  path = encodedir(path)
                  ef = _encodefname(path).split(b'/')
                  res = b'/'.join(_auxencode(ef, dotencode))
                  if len(res) > _maxstorepathlen:
                      res = _hashencode(path, dotencode)
                  return res
              def _pathencode(path):
                  de = encodedir(path)
                  if len(path) > _maxstorepathlen:
                      return _hashencode(de, True)
                  ef = _encodefname(de).split(b'/')
                  res = b'/'.join(_auxencode(ef, True))
                  if len(res) > _maxstorepathlen:
                      return _hashencode(de, True)
                  return res
              _pathencode = getattr(parsers, 'pathencode', _pathencode)
              def _plainhybridencode(f):
                  return _hybridencode(f, False)
              def _calcmode(vfs):
                  try:
                      # files in .hg/ will be created using this mode
                      mode = vfs.stat().st_mode
                      # avoid some useless chmods
                      if (0o777 & ~util.umask) == (0o777 & mode):
                          mode = None
                  except OSError:
                      mode = None
                  return mode
              _data = [
                  b'bookmarks',
                  b'narrowspec',
                  b'data',
                  b'meta',
                  b'00manifest.d',
                  b'00manifest.i',
                  b'00changelog.d',
                  b'00changelog.i',
                  b'phaseroots',
                  b'obsstore',
                  b'requires',
              ]
              REVLOG_FILES_MAIN_EXT = (b'.i',)
              REVLOG_FILES_OTHER_EXT = (
                  b'.idx',
                  b'.d',
                  b'.dat',
                  b'.n',
                  b'.nd',
                  b'.sda',
              )
              # file extension that also use a `-SOMELONGIDHASH.ext` form
              REVLOG_FILES_LONG_EXT = (
                  b'.nd',
                  b'.idx',
                  b'.dat',
                  b'.sda',
              )
              # files that are "volatile" and might change between listing and streaming
              #
              # note: the ".nd" file are nodemap data and won't "change" but they might be
              # deleted.
              REVLOG_FILES_VOLATILE_EXT = (b'.n', b'.nd')
              # some exception to the above matching
              #
              # XXX This is currently not in use because of issue6542
              EXCLUDED = re.compile(br'.*undo\.[^/]+\.(nd?|i)$')
              def is_revlog(f, kind, st):
                  if kind != stat.S_IFREG:
                      return None
                  return revlog_type(f)
              def revlog_type(f):
                  # XXX we need to filter `undo.` created by the transaction here, however
                  # being naive about it also filter revlog for `undo.*` files, leading to
                  # issue6542. So we no longer use EXCLUDED.
                  if f.endswith(REVLOG_FILES_MAIN_EXT):
                      return FILEFLAGS_REVLOG_MAIN
                  elif f.endswith(REVLOG_FILES_OTHER_EXT):
                      t = FILETYPE_FILELOG_OTHER
                      if f.endswith(REVLOG_FILES_VOLATILE_EXT):
                          t |= FILEFLAGS_VOLATILE
                      return t
                  return None
              # the file is part of changelog data
              FILEFLAGS_CHANGELOG = 1 << 13
              # the file is part of manifest data
              FILEFLAGS_MANIFESTLOG = 1 << 12
              # the file is part of filelog data
              FILEFLAGS_FILELOG = 1 << 11
              # file that are not directly part of a revlog
              FILEFLAGS_OTHER = 1 << 10
              # the main entry point for a revlog
              FILEFLAGS_REVLOG_MAIN = 1 << 1
              # a secondary file for a revlog
              FILEFLAGS_REVLOG_OTHER = 1 << 0
              # files that are "volatile" and might change between listing and streaming
              FILEFLAGS_VOLATILE = 1 << 20
              FILETYPE_CHANGELOG_MAIN = FILEFLAGS_CHANGELOG | FILEFLAGS_REVLOG_MAIN
              FILETYPE_CHANGELOG_OTHER = FILEFLAGS_CHANGELOG | FILEFLAGS_REVLOG_OTHER
              FILETYPE_MANIFESTLOG_MAIN = FILEFLAGS_MANIFESTLOG | FILEFLAGS_REVLOG_MAIN
              FILETYPE_MANIFESTLOG_OTHER = FILEFLAGS_MANIFESTLOG | FILEFLAGS_REVLOG_OTHER
              FILETYPE_FILELOG_MAIN = FILEFLAGS_FILELOG | FILEFLAGS_REVLOG_MAIN
              FILETYPE_FILELOG_OTHER = FILEFLAGS_FILELOG | FILEFLAGS_REVLOG_OTHER
              FILETYPE_OTHER = FILEFLAGS_OTHER
              @attr.s(slots=True, init=False)
              class BaseStoreEntry:
                  """An entry in the store
                  This is returned by `store.walk` and represent some data in the store."""
              @attr.s(slots=True, init=False)
              class SimpleStoreEntry(BaseStoreEntry):
                  """A generic entry in the store"""
                  is_revlog = False
                  _entry_path = attr.ib()
                  _is_volatile = attr.ib(default=False)
                  _file_size = attr.ib(default=None)
                  def __init__(
                      self,
                      entry_path,
                      is_volatile=False,
                      file_size=None,
                  ):
                      super().__init__()
                      self._entry_path = entry_path
                      self._is_volatile = is_volatile
                      self._file_size = file_size
                  def files(self):
                      return [
                          StoreFile(
                              unencoded_path=self._entry_path,
                              file_size=self._file_size,
                              is_volatile=self._is_volatile,
                          )
                      ]
              @attr.s(slots=True, init=False)
              class RevlogStoreEntry(BaseStoreEntry):
                  """A revlog entry in the store"""
                  is_revlog = True
                  revlog_type = attr.ib(default=None)
                  target_id = attr.ib(default=None)
                  _path_prefix = attr.ib(default=None)
                  _details = attr.ib(default=None)
                  def __init__(
                      self,
                      revlog_type,
                      path_prefix,
                      target_id,
                      details,
                  ):
                      super().__init__()
                      self.revlog_type = revlog_type
                      self.target_id = target_id
                      self._path_prefix = path_prefix
                      assert b'.i' in details, (path_prefix, details)
                      self._details = details
                  @property
                  def is_changelog(self):
                      return self.revlog_type & FILEFLAGS_CHANGELOG
                  @property
                  def is_manifestlog(self):
                      return self.revlog_type & FILEFLAGS_MANIFESTLOG
                  @property
                  def is_filelog(self):
                      return self.revlog_type & FILEFLAGS_FILELOG
                  def main_file_path(self):
                      """unencoded path of the main revlog file"""
                      return self._path_prefix + b'.i'
                  def files(self):
                      files = []
                      for ext in sorted(self._details, key=_ext_key):
                          path = self._path_prefix + ext
                          data = self._details[ext]
                          files.append(StoreFile(unencoded_path=path, **data))
                      return files
              @attr.s(slots=True)
              class StoreFile:
                  """a file matching an entry"""
                  unencoded_path = attr.ib()
                  _file_size = attr.ib(default=None)
                  is_volatile = attr.ib(default=False)
                  def file_size(self, vfs):
                      if self._file_size is not None:
                          return self._file_size
                      try:
                          return vfs.stat(self.unencoded_path).st_size
                      except FileNotFoundError:
                          return 0
              def _gather_revlog(files_data):
                  """group files per revlog prefix
                  The returns a two level nested dict. The top level key is the revlog prefix
                  without extension, the second level is all the file "suffix" that were
                  seen for this revlog and arbitrary file data as value.
                  """
                  revlogs = collections.defaultdict(dict)
                  for u, value in files_data:
                      name, ext = _split_revlog_ext(u)
                      revlogs[name][ext] = value
                  return sorted(revlogs.items())
              def _split_revlog_ext(filename):
                  """split the revlog file prefix from the variable extension"""
                  if filename.endswith(REVLOG_FILES_LONG_EXT):
                      char = b'-'
                  else:
                      char = b'.'
                  idx = filename.rfind(char)
                  return filename[:idx], filename[idx:]
              def _ext_key(ext):
                  """a key to order revlog suffix
                  important to issue .i after other entry."""
                  # the only important part of this order is to keep the `.i` last.
                  if ext.endswith(b'.n'):
                      return (0, ext)
                  elif ext.endswith(b'.nd'):
                      return (10, ext)
                  elif ext.endswith(b'.d'):
                      return (20, ext)
                  elif ext.endswith(b'.i'):
                      return (50, ext)
                  else:
                      return (40, ext)
              class basicstore:
                  '''base class for local repository stores'''
                  def __init__(self, path, vfstype):
                      vfs = vfstype(path)
                      self.path = vfs.base
                      self.createmode = _calcmode(vfs)
                      vfs.createmode = self.createmode
                      self.rawvfs = vfs
                      self.vfs = vfsmod.filtervfs(vfs, encodedir)
                      self.opener = self.vfs
                  def join(self, f):
                      return self.path + b'/' + encodedir(f)
                  def _walk(self, relpath, recurse, undecodable=None):
                      '''yields (revlog_type, unencoded, size)'''
                      path = self.path
                      if relpath:
                          path += b'/' + relpath
                      striplen = len(self.path) + 1
                      l = []
                      if self.rawvfs.isdir(path):
                          visit = [path]
                          readdir = self.rawvfs.readdir
                          while visit:
                              p = visit.pop()
                              for f, kind, st in readdir(p, stat=True):
                                  fp = p + b'/' + f
                                  rl_type = is_revlog(f, kind, st)
                                  if rl_type is not None:
                                      n = util.pconvert(fp[striplen:])
                                      l.append((decodedir(n), (rl_type, st.st_size)))
                                  elif kind == stat.S_IFDIR and recurse:
                                      visit.append(fp)
                      l.sort()
                      return l
                  def changelog(self, trypending, concurrencychecker=None):
                      return changelog.changelog(
                          self.vfs,
                          trypending=trypending,
                          concurrencychecker=concurrencychecker,
                      )
                  def manifestlog(self, repo, storenarrowmatch):
                      rootstore = manifest.manifestrevlog(repo.nodeconstants, self.vfs)
                      return manifest.manifestlog(self.vfs, repo, rootstore, storenarrowmatch)
-                 def datafiles(
+                 def data_entries(
                      self, matcher=None, undecodable=None
                  ) -> Generator[BaseStoreEntry, None, None]:
                      """Like walk, but excluding the changelog and root manifest.
                      When [undecodable] is None, revlogs names that can't be
                      decoded cause an exception. When it is provided, it should
                      be a list and the filenames that can't be decoded are added
                      to it instead. This is very rarely needed."""
                      dirs = [
                          (b'data', FILEFLAGS_FILELOG),
                          (b'meta', FILEFLAGS_MANIFESTLOG),
                      ]
                      for base_dir, rl_type in dirs:
                          files = self._walk(base_dir, True, undecodable=undecodable)
                          files = (f for f in files if f[1][0] is not None)
                          for revlog, details in _gather_revlog(files):
                              file_details = {}
                              revlog_target_id = revlog.split(b'/', 1)[1]
                              for ext, (t, s) in sorted(details.items()):
                                  file_details[ext] = {
                                      'is_volatile': bool(t & FILEFLAGS_VOLATILE),
                                      'file_size': s,
                                  }
                              yield RevlogStoreEntry(
                                  path_prefix=revlog,
                                  revlog_type=rl_type,
                                  target_id=revlog_target_id,
                                  details=file_details,
                              )
                  def topfiles(self) -> Generator[BaseStoreEntry, None, None]:
                      files = reversed(self._walk(b'', False))
                      changelogs = collections.defaultdict(dict)
                      manifestlogs = collections.defaultdict(dict)
                      for u, (t, s) in files:
                          if u.startswith(b'00changelog'):
                              name, ext = _split_revlog_ext(u)
                              changelogs[name][ext] = (t, s)
                          elif u.startswith(b'00manifest'):
                              name, ext = _split_revlog_ext(u)
                              manifestlogs[name][ext] = (t, s)
                          else:
                              yield SimpleStoreEntry(
                                  entry_path=u,
                                  is_volatile=bool(t & FILEFLAGS_VOLATILE),
                                  file_size=s,
                              )
                      # yield manifest before changelog
                      top_rl = [
                          (manifestlogs, FILEFLAGS_MANIFESTLOG),
                          (changelogs, FILEFLAGS_CHANGELOG),
                      ]
                      assert len(manifestlogs) <= 1
                      assert len(changelogs) <= 1
                      for data, revlog_type in top_rl:
                          for revlog, details in sorted(data.items()):
                              file_details = {}
                              for ext, (t, s) in details.items():
                                  file_details[ext] = {
                                      'is_volatile': bool(t & FILEFLAGS_VOLATILE),
                                      'file_size': s,
                                  }
                              yield RevlogStoreEntry(
                                  path_prefix=revlog,
                                  revlog_type=revlog_type,
                                  target_id=b'',
                                  details=file_details,
                              )
                  def walk(self, matcher=None) -> Generator[BaseStoreEntry, None, None]:
                      """return files related to data storage (ie: revlogs)
                      yields (file_type, unencoded, size)
                      if a matcher is passed, storage files of only those tracked paths
                      are passed with matches the matcher
                      """
                      # yield data files first
-                     for x in self.datafiles(matcher):
+                     for x in self.data_entries(matcher):
                          yield x
                      for x in self.topfiles():
                          yield x
                  def copylist(self):
                      return _data
                  def write(self, tr):
                      pass
                  def invalidatecaches(self):
                      pass
                  def markremoved(self, fn):
                      pass
                  def __contains__(self, path):
                      '''Checks if the store contains path'''
                      path = b"/".join((b"data", path))
                      # file?
                      if self.vfs.exists(path + b".i"):
                          return True
                      # dir?
                      if not path.endswith(b"/"):
                          path = path + b"/"
                      return self.vfs.exists(path)
              class encodedstore(basicstore):
                  def __init__(self, path, vfstype):
                      vfs = vfstype(path + b'/store')
                      self.path = vfs.base
                      self.createmode = _calcmode(vfs)
                      vfs.createmode = self.createmode
                      self.rawvfs = vfs
                      self.vfs = vfsmod.filtervfs(vfs, encodefilename)
                      self.opener = self.vfs
                  def _walk(self, relpath, recurse, undecodable=None):
                      old = super()._walk(relpath, recurse)
                      new = []
                      for f1, value in old:
                          try:
                              f2 = decodefilename(f1)
                          except KeyError:
                              if undecodable is None:
                                  msg = _(b'undecodable revlog name %s') % f1
                                  raise error.StorageError(msg)
                              else:
                                  undecodable.append(f1)
                                  continue
                          new.append((f2, value))
                      return new
-                 def datafiles(
+                 def data_entries(
                      self, matcher=None, undecodable=None
                  ) -> Generator[BaseStoreEntry, None, None]:
-                     entries = super(encodedstore, self).datafiles(undecodable=undecodable)
+                     entries = super(encodedstore, self).data_entries(
+                         undecodable=undecodable
+                     )
                      for entry in entries:
                          if _match_tracked_entry(entry, matcher):
                              yield entry
                  def join(self, f):
                      return self.path + b'/' + encodefilename(f)
                  def copylist(self):
                      return [b'requires', b'00changelog.i'] + [b'store/' + f for f in _data]
              class fncache:
                  # the filename used to be partially encoded
                  # hence the encodedir/decodedir dance
                  def __init__(self, vfs):
                      self.vfs = vfs
                      self._ignores = set()
                      self.entries = None
                      self._dirty = False
                      # set of new additions to fncache
                      self.addls = set()
                  def ensureloaded(self, warn=None):
                      """read the fncache file if not already read.
                      If the file on disk is corrupted, raise. If warn is provided,
                      warn and keep going instead."""
                      if self.entries is None:
                          self._load(warn)
                  def _load(self, warn=None):
                      '''fill the entries from the fncache file'''
                      self._dirty = False
                      try:
                          fp = self.vfs(b'fncache', mode=b'rb')
                      except IOError:
                          # skip nonexistent file
                          self.entries = set()
                          return
                      self.entries = set()
                      chunk = b''
                      for c in iter(functools.partial(fp.read, fncache_chunksize), b''):
                          chunk += c
                          try:
                              p = chunk.rindex(b'\n')
                              self.entries.update(decodedir(chunk[: p + 1]).splitlines())
                              chunk = chunk[p + 1 :]
                          except ValueError:
                              # substring '\n' not found, maybe the entry is bigger than the
                              # chunksize, so let's keep iterating
                              pass
                      if chunk:
                          msg = _(b"fncache does not ends with a newline")
                          if warn:
                              warn(msg + b'\n')
                          else:
                              raise error.Abort(
                                  msg,
                                  hint=_(
                                      b"use 'hg debugrebuildfncache' to "
                                      b"rebuild the fncache"
                                  ),
                              )
                      self._checkentries(fp, warn)
                      fp.close()
                  def _checkentries(self, fp, warn):
                      """make sure there is no empty string in entries"""
                      if b'' in self.entries:
                          fp.seek(0)
                          for n, line in enumerate(fp):
                              if not line.rstrip(b'\n'):
                                  t = _(b'invalid entry in fncache, line %d') % (n + 1)
                                  if warn:
                                      warn(t + b'\n')
                                  else:
                                      raise error.Abort(t)
                  def write(self, tr):
                      if self._dirty:
                          assert self.entries is not None
                          self.entries = self.entries | self.addls
                          self.addls = set()
                          tr.addbackup(b'fncache')
                          fp = self.vfs(b'fncache', mode=b'wb', atomictemp=True)
                          if self.entries:
                              fp.write(encodedir(b'\n'.join(self.entries) + b'\n'))
                          fp.close()
                          self._dirty = False
                      if self.addls:
                          # if we have just new entries, let's append them to the fncache
                          tr.addbackup(b'fncache')
                          fp = self.vfs(b'fncache', mode=b'ab', atomictemp=True)
                          if self.addls:
                              fp.write(encodedir(b'\n'.join(self.addls) + b'\n'))
                          fp.close()
                          self.entries = None
                          self.addls = set()
                  def addignore(self, fn):
                      self._ignores.add(fn)
                  def add(self, fn):
                      if fn in self._ignores:
                          return
                      if self.entries is None:
                          self._load()
                      if fn not in self.entries:
                          self.addls.add(fn)
                  def remove(self, fn):
                      if self.entries is None:
                          self._load()
                      if fn in self.addls:
                          self.addls.remove(fn)
                          return
                      try:
                          self.entries.remove(fn)
                          self._dirty = True
                      except KeyError:
                          pass
                  def __contains__(self, fn):
                      if fn in self.addls:
                          return True
                      if self.entries is None:
                          self._load()
                      return fn in self.entries
                  def __iter__(self):
                      if self.entries is None:
                          self._load()
                      return iter(self.entries | self.addls)
              class _fncachevfs(vfsmod.proxyvfs):
                  def __init__(self, vfs, fnc, encode):
                      vfsmod.proxyvfs.__init__(self, vfs)
                      self.fncache = fnc
                      self.encode = encode
                  def __call__(self, path, mode=b'r', *args, **kw):
                      encoded = self.encode(path)
                      if (
                          mode not in (b'r', b'rb')
                          and (path.startswith(b'data/') or path.startswith(b'meta/'))
                          and revlog_type(path) is not None
                      ):
                          # do not trigger a fncache load when adding a file that already is
                          # known to exist.
                          notload = self.fncache.entries is None and self.vfs.exists(encoded)
                          if notload and b'r+' in mode and not self.vfs.stat(encoded).st_size:
                              # when appending to an existing file, if the file has size zero,
                              # it should be considered as missing. Such zero-size files are
                              # the result of truncation when a transaction is aborted.
                              notload = False
                          if not notload:
                              self.fncache.add(path)
                      return self.vfs(encoded, mode, *args, **kw)
                  def join(self, path):
                      if path:
                          return self.vfs.join(self.encode(path))
                      else:
                          return self.vfs.join(path)
                  def register_file(self, path):
                      """generic hook point to lets fncache steer its stew"""
                      if path.startswith(b'data/') or path.startswith(b'meta/'):
                          self.fncache.add(path)
              class fncachestore(basicstore):
                  def __init__(self, path, vfstype, dotencode):
                      if dotencode:
                          encode = _pathencode
                      else:
                          encode = _plainhybridencode
                      self.encode = encode
                      vfs = vfstype(path + b'/store')
                      self.path = vfs.base
                      self.pathsep = self.path + b'/'
                      self.createmode = _calcmode(vfs)
                      vfs.createmode = self.createmode
                      self.rawvfs = vfs
                      fnc = fncache(vfs)
                      self.fncache = fnc
                      self.vfs = _fncachevfs(vfs, fnc, encode)
                      self.opener = self.vfs
                  def join(self, f):
                      return self.pathsep + self.encode(f)
                  def getsize(self, path):
                      return self.rawvfs.stat(path).st_size
-                 def datafiles(
+                 def data_entries(
                      self, matcher=None, undecodable=None
                  ) -> Generator[BaseStoreEntry, None, None]:
                      files = ((f, revlog_type(f)) for f in self.fncache)
                      # Note: all files in fncache should be revlog related, However the
                      # fncache might contains such file added by previous version of
                      # Mercurial.
                      files = (f for f in files if f[1] is not None)
                      by_revlog = _gather_revlog(files)
                      for revlog, details in by_revlog:
                          file_details = {}
                          if revlog.startswith(b'data/'):
                              rl_type = FILEFLAGS_FILELOG
                              revlog_target_id = revlog.split(b'/', 1)[1]
                          elif revlog.startswith(b'meta/'):
                              rl_type = FILEFLAGS_MANIFESTLOG
                              # drop the initial directory and the `00manifest` file part
                              tmp = revlog.split(b'/', 1)[1]
                              revlog_target_id = tmp.rsplit(b'/', 1)[0] + b'/'
                          else:
                              # unreachable
                              assert False, revlog
                          for ext, t in details.items():
                              file_details[ext] = {
                                  'is_volatile': bool(t & FILEFLAGS_VOLATILE),
                              }
                          entry = RevlogStoreEntry(
                              path_prefix=revlog,
                              revlog_type=rl_type,
                              target_id=revlog_target_id,
                              details=file_details,
                          )
                          if _match_tracked_entry(entry, matcher):
                              yield entry
                  def copylist(self):
                      d = (
                          b'bookmarks',
                          b'narrowspec',
                          b'data',
                          b'meta',
                          b'dh',
                          b'fncache',
                          b'phaseroots',
                          b'obsstore',
                          b'00manifest.d',
                          b'00manifest.i',
                          b'00changelog.d',
                          b'00changelog.i',
                          b'requires',
                      )
                      return [b'requires', b'00changelog.i'] + [b'store/' + f for f in d]
                  def write(self, tr):
                      self.fncache.write(tr)
                  def invalidatecaches(self):
                      self.fncache.entries = None
                      self.fncache.addls = set()
                  def markremoved(self, fn):
                      self.fncache.remove(fn)
                  def _exists(self, f):
                      ef = self.encode(f)
                      try:
                          self.getsize(ef)
                          return True
                      except FileNotFoundError:
                          return False
                  def __contains__(self, path):
                      '''Checks if the store contains path'''
                      path = b"/".join((b"data", path))
                      # check for files (exact match)
                      e = path + b'.i'
                      if e in self.fncache and self._exists(e):
                          return True
                      # now check for directories (prefix match)
                      if not path.endswith(b'/'):
                          path += b'/'
                      for e in self.fncache:
                          if e.startswith(path) and self._exists(e):
                              return True
                      return False

mercurial/verify.py

0 +2 -2

              # verify.py - repository integrity checking for Mercurial
              #
              # Copyright 2006, 2007 Olivia Mackall <olivia@selenic.com>
              #
              # This software may be used and distributed according to the terms of the
              # GNU General Public License version 2 or any later version.
              import os
              from .i18n import _
              from .node import short
              from .utils import stringutil
              from . import (
                  error,
                  pycompat,
                  requirements,
                  revlog,
                  util,
              )
              VERIFY_DEFAULT = 0
              VERIFY_FULL = 1
              def verify(repo, level=None):
                  with repo.lock():
                      v = verifier(repo, level)
                      return v.verify()
              def _normpath(f):
                  # under hg < 2.4, convert didn't sanitize paths properly, so a
                  # converted repo may contain repeated slashes
                  while b'//' in f:
                      f = f.replace(b'//', b'/')
                  return f
              HINT_FNCACHE = _(
                  b'hint: run "hg debugrebuildfncache" to recover from corrupt fncache\n'
              )
              WARN_PARENT_DIR_UNKNOWN_REV = _(
                  b"parent-directory manifest refers to unknown revision %s"
              )
              WARN_UNKNOWN_COPY_SOURCE = _(
                  b"warning: copy source of '%s' not in parents of %s"
              )
              WARN_NULLID_COPY_SOURCE = _(
                  b"warning: %s@%s: copy source revision is nullid %s:%s\n"
              )
              class verifier:
                  def __init__(self, repo, level=None):
                      self.repo = repo.unfiltered()
                      self.ui = repo.ui
                      self.match = repo.narrowmatch()
                      if level is None:
                          level = VERIFY_DEFAULT
                      self._level = level
                      self.badrevs = set()
                      self.errors = 0
                      self.warnings = 0
                      self.havecl = len(repo.changelog) > 0
                      self.havemf = len(repo.manifestlog.getstorage(b'')) > 0
                      self.revlogv1 = repo.changelog._format_version != revlog.REVLOGV0
                      self.lrugetctx = util.lrucachefunc(repo.unfiltered().__getitem__)
                      self.refersmf = False
                      self.fncachewarned = False
                      # developer config: verify.skipflags
                      self.skipflags = repo.ui.configint(b'verify', b'skipflags')
                      self.warnorphanstorefiles = True
                  def _warn(self, msg):
                      """record a "warning" level issue"""
                      self.ui.warn(msg + b"\n")
                      self.warnings += 1
                  def _err(self, linkrev, msg, filename=None):
                      """record a "error" level issue"""
                      if linkrev is not None:
                          self.badrevs.add(linkrev)
                          linkrev = b"%d" % linkrev
                      else:
                          linkrev = b'?'
                      msg = b"%s: %s" % (linkrev, msg)
                      if filename:
                          msg = b"%s@%s" % (filename, msg)
                      self.ui.warn(b" " + msg + b"\n")
                      self.errors += 1
                  def _exc(self, linkrev, msg, inst, filename=None):
                      """record exception raised during the verify process"""
                      fmsg = stringutil.forcebytestr(inst)
                      if not fmsg:
                          fmsg = pycompat.byterepr(inst)
                      self._err(linkrev, b"%s: %s" % (msg, fmsg), filename)
                  def _checkrevlog(self, obj, name, linkrev):
                      """verify high level property of a revlog
                      - revlog is present,
                      - revlog is non-empty,
                      - sizes (index and data) are correct,
                      - revlog's format version is correct.
                      """
                      if not len(obj) and (self.havecl or self.havemf):
                          self._err(linkrev, _(b"empty or missing %s") % name)
                          return
                      d = obj.checksize()
                      if d[0]:
                          self._err(None, _(b"data length off by %d bytes") % d[0], name)
                      if d[1]:
                          self._err(None, _(b"index contains %d extra bytes") % d[1], name)
                      if obj._format_version != revlog.REVLOGV0:
                          if not self.revlogv1:
                              self._warn(_(b"warning: `%s' uses revlog format 1") % name)
                      elif self.revlogv1:
                          self._warn(_(b"warning: `%s' uses revlog format 0") % name)
                  def _checkentry(self, obj, i, node, seen, linkrevs, f):
                      """verify a single revlog entry
                      arguments are:
                      - obj:      the source revlog
                      - i:        the revision number
                      - node:     the revision node id
                      - seen:     nodes previously seen for this revlog
                      - linkrevs: [changelog-revisions] introducing "node"
                      - f:        string label ("changelog", "manifest", or filename)
                      Performs the following checks:
                      - linkrev points to an existing changelog revision,
                      - linkrev points to a changelog revision that introduces this revision,
                      - linkrev points to the lowest of these changesets,
                      - both parents exist in the revlog,
                      - the revision is not duplicated.
                      Return the linkrev of the revision (or None for changelog's revisions).
                      """
                      lr = obj.linkrev(obj.rev(node))
                      if lr < 0 or (self.havecl and lr not in linkrevs):
                          if lr < 0 or lr >= len(self.repo.changelog):
                              msg = _(b"rev %d points to nonexistent changeset %d")
                          else:
                              msg = _(b"rev %d points to unexpected changeset %d")
                          self._err(None, msg % (i, lr), f)
                          if linkrevs:
                              if f and len(linkrevs) > 1:
                                  try:
                                      # attempt to filter down to real linkrevs
                                      linkrevs = []
                                      for lr in linkrevs:
                                          if self.lrugetctx(lr)[f].filenode() == node:
                                              linkrevs.append(lr)
                                  except Exception:
                                      pass
                              msg = _(b" (expected %s)")
                              msg %= b" ".join(map(pycompat.bytestr, linkrevs))
                              self._warn(msg)
                          lr = None  # can't be trusted
                      try:
                          p1, p2 = obj.parents(node)
                          if p1 not in seen and p1 != self.repo.nullid:
                              msg = _(b"unknown parent 1 %s of %s") % (short(p1), short(node))
                              self._err(lr, msg, f)
                          if p2 not in seen and p2 != self.repo.nullid:
                              msg = _(b"unknown parent 2 %s of %s") % (short(p2), short(node))
                              self._err(lr, msg, f)
                      except Exception as inst:
                          self._exc(lr, _(b"checking parents of %s") % short(node), inst, f)
                      if node in seen:
                          self._err(lr, _(b"duplicate revision %d (%d)") % (i, seen[node]), f)
                      seen[node] = i
                      return lr
                  def verify(self):
                      """verify the content of the Mercurial repository
                      This method run all verifications, displaying issues as they are found.
                      return 1 if any error have been encountered, 0 otherwise."""
                      # initial validation and generic report
                      repo = self.repo
                      ui = repo.ui
                      if not repo.url().startswith(b'file:'):
                          raise error.Abort(_(b"cannot verify bundle or remote repos"))
                      if os.path.exists(repo.sjoin(b"journal")):
                          ui.warn(_(b"abandoned transaction found - run hg recover\n"))
                      if ui.verbose or not self.revlogv1:
                          ui.status(
                              _(b"repository uses revlog format %d\n")
                              % (self.revlogv1 and 1 or 0)
                          )
                      # data verification
                      mflinkrevs, filelinkrevs = self._verifychangelog()
                      filenodes = self._verifymanifest(mflinkrevs)
                      del mflinkrevs
                      self._crosscheckfiles(filelinkrevs, filenodes)
                      totalfiles, filerevisions = self._verifyfiles(filenodes, filelinkrevs)
                      if self.errors:
                          ui.warn(_(b"not checking dirstate because of previous errors\n"))
                          dirstate_errors = 0
                      else:
                          dirstate_errors = self._verify_dirstate()
                      # final report
                      ui.status(
                          _(b"checked %d changesets with %d changes to %d files\n")
                          % (len(repo.changelog), filerevisions, totalfiles)
                      )
                      if self.warnings:
                          ui.warn(_(b"%d warnings encountered!\n") % self.warnings)
                      if self.fncachewarned:
                          ui.warn(HINT_FNCACHE)
                      if self.errors:
                          ui.warn(_(b"%d integrity errors encountered!\n") % self.errors)
                          if self.badrevs:
                              msg = _(b"(first damaged changeset appears to be %d)\n")
                              msg %= min(self.badrevs)
                              ui.warn(msg)
                          if dirstate_errors:
                              ui.warn(
                                  _(b"dirstate inconsistent with current parent's manifest\n")
                              )
                              ui.warn(_(b"%d dirstate errors\n") % dirstate_errors)
                          return 1
                      return 0
                  def _verifychangelog(self):
                      """verify the changelog of a repository
                      The following checks are performed:
                      - all of `_checkrevlog` checks,
                      - all of `_checkentry` checks (for each revisions),
                      - each revision can be read.
                      The function returns some of the data observed in the changesets as a
                      (mflinkrevs, filelinkrevs) tuples:
                      - mflinkrevs:   is a { manifest-node -> [changelog-rev] } mapping
                      - filelinkrevs: is a { file-path -> [changelog-rev] } mapping
                      If a matcher was specified, filelinkrevs will only contains matched
                      files.
                      """
                      ui = self.ui
                      repo = self.repo
                      match = self.match
                      cl = repo.changelog
                      ui.status(_(b"checking changesets\n"))
                      mflinkrevs = {}
                      filelinkrevs = {}
                      seen = {}
                      self._checkrevlog(cl, b"changelog", 0)
                      progress = ui.makeprogress(
                          _(b'checking'), unit=_(b'changesets'), total=len(repo)
                      )
                      for i in repo:
                          progress.update(i)
                          n = cl.node(i)
                          self._checkentry(cl, i, n, seen, [i], b"changelog")
                          try:
                              changes = cl.read(n)
                              if changes[0] != self.repo.nullid:
                                  mflinkrevs.setdefault(changes[0], []).append(i)
                                  self.refersmf = True
                              for f in changes[3]:
                                  if match(f):
                                      filelinkrevs.setdefault(_normpath(f), []).append(i)
                          except Exception as inst:
                              self.refersmf = True
                              self._exc(i, _(b"unpacking changeset %s") % short(n), inst)
                      progress.complete()
                      return mflinkrevs, filelinkrevs
                  def _verifymanifest(
                      self, mflinkrevs, dir=b"", storefiles=None, subdirprogress=None
                  ):
                      """verify the manifestlog content
                      Inputs:
                      - mflinkrevs:     a {manifest-node -> [changelog-revisions]} mapping
                      - dir:            a subdirectory to check (for tree manifest repo)
                      - storefiles:     set of currently "orphan" files.
                      - subdirprogress: a progress object
                      This function checks:
                      * all of `_checkrevlog` checks (for all manifest related revlogs)
                      * all of `_checkentry` checks (for all manifest related revisions)
                      * nodes for subdirectory exists in the sub-directory manifest
                      * each manifest entries have a file path
                      * each manifest node refered in mflinkrevs exist in the manifest log
                      If tree manifest is in use and a matchers is specified, only the
                      sub-directories matching it will be verified.
                      return a two level mapping:
                          {"path" -> { filenode -> changelog-revision}}
                      This mapping primarily contains entries for every files in the
                      repository. In addition, when tree-manifest is used, it also contains
                      sub-directory entries.
                      If a matcher is provided, only matching paths will be included.
                      """
                      repo = self.repo
                      ui = self.ui
                      match = self.match
                      mfl = self.repo.manifestlog
                      mf = mfl.getstorage(dir)
                      if not dir:
                          self.ui.status(_(b"checking manifests\n"))
                      filenodes = {}
                      subdirnodes = {}
                      seen = {}
                      label = b"manifest"
                      if dir:
                          label = dir
                          revlogfiles = mf.files()
                          storefiles.difference_update(revlogfiles)
                          if subdirprogress:  # should be true since we're in a subdirectory
                              subdirprogress.increment()
                      if self.refersmf:
                          # Do not check manifest if there are only changelog entries with
                          # null manifests.
                          self._checkrevlog(mf._revlog, label, 0)
                      progress = ui.makeprogress(
                          _(b'checking'), unit=_(b'manifests'), total=len(mf)
                      )
                      for i in mf:
                          if not dir:
                              progress.update(i)
                          n = mf.node(i)
                          lr = self._checkentry(mf, i, n, seen, mflinkrevs.get(n, []), label)
                          if n in mflinkrevs:
                              del mflinkrevs[n]
                          elif dir:
                              msg = _(b"%s not in parent-directory manifest") % short(n)
                              self._err(lr, msg, label)
                          else:
                              self._err(lr, _(b"%s not in changesets") % short(n), label)
                          try:
                              mfdelta = mfl.get(dir, n).readdelta(shallow=True)
                              for f, fn, fl in mfdelta.iterentries():
                                  if not f:
                                      self._err(lr, _(b"entry without name in manifest"))
                                  elif f == b"/dev/null":  # ignore this in very old repos
                                      continue
                                  fullpath = dir + _normpath(f)
                                  if fl == b't':
                                      if not match.visitdir(fullpath):
                                          continue
                                      sdn = subdirnodes.setdefault(fullpath + b'/', {})
                                      sdn.setdefault(fn, []).append(lr)
                                  else:
                                      if not match(fullpath):
                                          continue
                                      filenodes.setdefault(fullpath, {}).setdefault(fn, lr)
                          except Exception as inst:
                              self._exc(lr, _(b"reading delta %s") % short(n), inst, label)
                          if self._level >= VERIFY_FULL:
                              try:
                                  # Various issues can affect manifest. So we read each full
                                  # text from storage. This triggers the checks from the core
                                  # code (eg: hash verification, filename are ordered, etc.)
                                  mfdelta = mfl.get(dir, n).read()
                              except Exception as inst:
                                  msg = _(b"reading full manifest %s") % short(n)
                                  self._exc(lr, msg, inst, label)
                      if not dir:
                          progress.complete()
                      if self.havemf:
                          # since we delete entry in `mflinkrevs` during iteration, any
                          # remaining entries are "missing". We need to issue errors for them.
                          changesetpairs = [(c, m) for m in mflinkrevs for c in mflinkrevs[m]]
                          for c, m in sorted(changesetpairs):
                              if dir:
                                  self._err(c, WARN_PARENT_DIR_UNKNOWN_REV % short(m), label)
                              else:
                                  msg = _(b"changeset refers to unknown revision %s")
                                  msg %= short(m)
                                  self._err(c, msg, label)
                      if not dir and subdirnodes:
                          self.ui.status(_(b"checking directory manifests\n"))
                          storefiles = set()
                          subdirs = set()
                          revlogv1 = self.revlogv1
                          undecodable = []
-                         for entry in repo.store.datafiles(undecodable=undecodable):
+                         for entry in repo.store.data_entries(undecodable=undecodable):
                              for file_ in entry.files():
                                  f = file_.unencoded_path
                                  size = file_.file_size(repo.store.vfs)
                                  if (size > 0 or not revlogv1) and f.startswith(b'meta/'):
                                      storefiles.add(_normpath(f))
                                      subdirs.add(os.path.dirname(f))
                          for f in undecodable:
                              self._err(None, _(b"cannot decode filename '%s'") % f)
                          subdirprogress = ui.makeprogress(
                              _(b'checking'), unit=_(b'manifests'), total=len(subdirs)
                          )
                      for subdir, linkrevs in subdirnodes.items():
                          subdirfilenodes = self._verifymanifest(
                              linkrevs, subdir, storefiles, subdirprogress
                          )
                          for f, onefilenodes in subdirfilenodes.items():
                              filenodes.setdefault(f, {}).update(onefilenodes)
                      if not dir and subdirnodes:
                          assert subdirprogress is not None  # help pytype
                          subdirprogress.complete()
                          if self.warnorphanstorefiles:
                              for f in sorted(storefiles):
                                  self._warn(_(b"warning: orphan data file '%s'") % f)
                      return filenodes
                  def _crosscheckfiles(self, filelinkrevs, filenodes):
                      repo = self.repo
                      ui = self.ui
                      ui.status(_(b"crosschecking files in changesets and manifests\n"))
                      total = len(filelinkrevs) + len(filenodes)
                      progress = ui.makeprogress(
                          _(b'crosschecking'), unit=_(b'files'), total=total
                      )
                      if self.havemf:
                          for f in sorted(filelinkrevs):
                              progress.increment()
                              if f not in filenodes:
                                  lr = filelinkrevs[f][0]
                                  self._err(lr, _(b"in changeset but not in manifest"), f)
                      if self.havecl:
                          for f in sorted(filenodes):
                              progress.increment()
                              if f not in filelinkrevs:
                                  try:
                                      fl = repo.file(f)
                                      lr = min([fl.linkrev(fl.rev(n)) for n in filenodes[f]])
                                  except Exception:
                                      lr = None
                                  self._err(lr, _(b"in manifest but not in changeset"), f)
                      progress.complete()
                  def _verifyfiles(self, filenodes, filelinkrevs):
                      repo = self.repo
                      ui = self.ui
                      lrugetctx = self.lrugetctx
                      revlogv1 = self.revlogv1
                      havemf = self.havemf
                      ui.status(_(b"checking files\n"))
                      storefiles = set()
                      undecodable = []
-                     for entry in repo.store.datafiles(undecodable=undecodable):
+                     for entry in repo.store.data_entries(undecodable=undecodable):
                          for file_ in entry.files():
                              size = file_.file_size(repo.store.vfs)
                              f = file_.unencoded_path
                              if (size > 0 or not revlogv1) and f.startswith(b'data/'):
                                  storefiles.add(_normpath(f))
                      for f in undecodable:
                          self._err(None, _(b"cannot decode filename '%s'") % f)
                      state = {
                          # TODO this assumes revlog storage for changelog.
                          b'expectedversion': self.repo.changelog._format_version,
                          b'skipflags': self.skipflags,
                          # experimental config: censor.policy
                          b'erroroncensored': ui.config(b'censor', b'policy') == b'abort',
                      }
                      files = sorted(set(filenodes) | set(filelinkrevs))
                      revisions = 0
                      progress = ui.makeprogress(
                          _(b'checking'), unit=_(b'files'), total=len(files)
                      )
                      for i, f in enumerate(files):
                          progress.update(i, item=f)
                          try:
                              linkrevs = filelinkrevs[f]
                          except KeyError:
                              # in manifest but not in changelog
                              linkrevs = []
                          if linkrevs:
                              lr = linkrevs[0]
                          else:
                              lr = None
                          try:
                              fl = repo.file(f)
                          except error.StorageError as e:
                              self._err(lr, _(b"broken revlog! (%s)") % e, f)
                              continue
                          for ff in fl.files():
                              try:
                                  storefiles.remove(ff)
                              except KeyError:
                                  if self.warnorphanstorefiles:
                                      msg = _(b" warning: revlog '%s' not in fncache!")
                                      self._warn(msg % ff)
                                      self.fncachewarned = True
                          if not len(fl) and (self.havecl or self.havemf):
                              self._err(lr, _(b"empty or missing %s") % f)
                          else:
                              # Guard against implementations not setting this.
                              state[b'skipread'] = set()
                              state[b'safe_renamed'] = set()
                              for problem in fl.verifyintegrity(state):
                                  if problem.node is not None:
                                      linkrev = fl.linkrev(fl.rev(problem.node))
                                  else:
                                      linkrev = None
                                  if problem.warning:
                                      self._warn(problem.warning)
                                  elif problem.error:
                                      linkrev_msg = linkrev if linkrev is not None else lr
                                      self._err(linkrev_msg, problem.error, f)
                                  else:
                                      raise error.ProgrammingError(
                                          b'problem instance does not set warning or error '
                                          b'attribute: %s' % problem.msg
                                      )
                          seen = {}
                          for i in fl:
                              revisions += 1
                              n = fl.node(i)
                              lr = self._checkentry(fl, i, n, seen, linkrevs, f)
                              if f in filenodes:
                                  if havemf and n not in filenodes[f]:
                                      self._err(lr, _(b"%s not in manifests") % (short(n)), f)
                                  else:
                                      del filenodes[f][n]
                              if n in state[b'skipread'] and n not in state[b'safe_renamed']:
                                  continue
                              # check renames
                              try:
                                  # This requires resolving fulltext (at least on revlogs,
                                  # though not with LFS revisions). We may want
                                  # ``verifyintegrity()`` to pass a set of nodes with
                                  # rename metadata as an optimization.
                                  rp = fl.renamed(n)
                                  if rp:
                                      if lr is not None and ui.verbose:
                                          ctx = lrugetctx(lr)
                                          if not any(rp[0] in pctx for pctx in ctx.parents()):
                                              self._warn(WARN_UNKNOWN_COPY_SOURCE % (f, ctx))
                                      fl2 = repo.file(rp[0])
                                      if not len(fl2):
                                          m = _(b"empty or missing copy source revlog %s:%s")
                                          self._err(lr, m % (rp[0], short(rp[1])), f)
                                      elif rp[1] == self.repo.nullid:
                                          msg = WARN_NULLID_COPY_SOURCE
                                          msg %= (f, lr, rp[0], short(rp[1]))
                                          ui.note(msg)
                                      else:
                                          fl2.rev(rp[1])
                              except Exception as inst:
                                  self._exc(
                                      lr, _(b"checking rename of %s") % short(n), inst, f
                                  )
                          # cross-check
                          if f in filenodes:
                              fns = [(v, k) for k, v in filenodes[f].items()]
                              for lr, node in sorted(fns):
                                  msg = _(b"manifest refers to unknown revision %s")
                                  self._err(lr, msg % short(node), f)
                      progress.complete()
                      if self.warnorphanstorefiles:
                          for f in sorted(storefiles):
                              self._warn(_(b"warning: orphan data file '%s'") % f)
                      return len(files), revisions
                  def _verify_dirstate(self):
                      """Check that the dirstate is consistent with the parent's manifest"""
                      repo = self.repo
                      ui = self.ui
                      ui.status(_(b"checking dirstate\n"))
                      parent1, parent2 = repo.dirstate.parents()
                      m1 = repo[parent1].manifest()
                      m2 = repo[parent2].manifest()
                      dirstate_errors = 0
                      is_narrow = requirements.NARROW_REQUIREMENT in repo.requirements
                      narrow_matcher = repo.narrowmatch() if is_narrow else None
                      for err in repo.dirstate.verify(m1, m2, parent1, narrow_matcher):
                          ui.error(err)
                          dirstate_errors += 1
                      if dirstate_errors:
                          self.errors += dirstate_errors
                      return dirstate_errors

tests/simplestorerepo.py

0 +2 -2

              # simplestorerepo.py - Extension that swaps in alternate repository storage.
              #
              # Copyright 2018 Gregory Szorc <gregory.szorc@gmail.com>
              #
              # This software may be used and distributed according to the terms of the
              # GNU General Public License version 2 or any later version.
              # To use this with the test suite:
              #
              #   $ HGREPOFEATURES="simplestore" ./run-tests.py \
              #       --extra-config-opt extensions.simplestore=`pwd`/simplestorerepo.py
              import stat
              from mercurial.i18n import _
              from mercurial.node import (
                  bin,
                  hex,
                  nullrev,
              )
              from mercurial.thirdparty import attr
              from mercurial import (
                  ancestor,
                  bundlerepo,
                  error,
                  extensions,
                  localrepo,
                  mdiff,
                  pycompat,
                  revlog,
                  store,
                  verify,
              )
              from mercurial.interfaces import (
                  repository,
                  util as interfaceutil,
              )
              from mercurial.utils import (
                  cborutil,
                  storageutil,
              )
              from mercurial.revlogutils import flagutil
              # Note for extension authors: ONLY specify testedwith = 'ships-with-hg-core' for
              # extensions which SHIP WITH MERCURIAL. Non-mainline extensions should
              # be specifying the version(s) of Mercurial they are tested with, or
              # leave the attribute unspecified.
              testedwith = b'ships-with-hg-core'
              REQUIREMENT = b'testonly-simplestore'
              def validatenode(node):
                  if isinstance(node, int):
                      raise ValueError('expected node; got int')
                  if len(node) != 20:
                      raise ValueError('expected 20 byte node')
              def validaterev(rev):
                  if not isinstance(rev, int):
                      raise ValueError('expected int')
              class simplestoreerror(error.StorageError):
                  pass
              @interfaceutil.implementer(repository.irevisiondelta)
              @attr.s(slots=True)
              class simplestorerevisiondelta:
                  node = attr.ib()
                  p1node = attr.ib()
                  p2node = attr.ib()
                  basenode = attr.ib()
                  flags = attr.ib()
                  baserevisionsize = attr.ib()
                  revision = attr.ib()
                  delta = attr.ib()
                  linknode = attr.ib(default=None)
              @interfaceutil.implementer(repository.iverifyproblem)
              @attr.s(frozen=True)
              class simplefilestoreproblem:
                  warning = attr.ib(default=None)
                  error = attr.ib(default=None)
                  node = attr.ib(default=None)
              @interfaceutil.implementer(repository.ifilestorage)
              class filestorage:
                  """Implements storage for a tracked path.
                  Data is stored in the VFS in a directory corresponding to the tracked
                  path.
                  Index data is stored in an ``index`` file using CBOR.
                  Fulltext data is stored in files having names of the node.
                  """
                  _flagserrorclass = simplestoreerror
                  def __init__(self, repo, svfs, path):
                      self.nullid = repo.nullid
                      self._repo = repo
                      self._svfs = svfs
                      self._path = path
                      self._storepath = b'/'.join([b'data', path])
                      self._indexpath = b'/'.join([self._storepath, b'index'])
                      indexdata = self._svfs.tryread(self._indexpath)
                      if indexdata:
                          indexdata = cborutil.decodeall(indexdata)
                      self._indexdata = indexdata or []
                      self._indexbynode = {}
                      self._indexbyrev = {}
                      self._index = []
                      self._refreshindex()
                      self._flagprocessors = dict(flagutil.flagprocessors)
                  def _refreshindex(self):
                      self._indexbynode.clear()
                      self._indexbyrev.clear()
                      self._index = []
                      for i, entry in enumerate(self._indexdata):
                          self._indexbynode[entry[b'node']] = entry
                          self._indexbyrev[i] = entry
                      self._indexbynode[self._repo.nullid] = {
                          b'node': self._repo.nullid,
                          b'p1': self._repo.nullid,
                          b'p2': self._repo.nullid,
                          b'linkrev': nullrev,
                          b'flags': 0,
                      }
                      self._indexbyrev[nullrev] = {
                          b'node': self._repo.nullid,
                          b'p1': self._repo.nullid,
                          b'p2': self._repo.nullid,
                          b'linkrev': nullrev,
                          b'flags': 0,
                      }
                      for i, entry in enumerate(self._indexdata):
                          p1rev, p2rev = self.parentrevs(self.rev(entry[b'node']))
                          # start, length, rawsize, chainbase, linkrev, p1, p2, node
                          self._index.append(
                              (0, 0, 0, -1, entry[b'linkrev'], p1rev, p2rev, entry[b'node'])
                          )
                      self._index.append((0, 0, 0, -1, -1, -1, -1, self._repo.nullid))
                  def __len__(self):
                      return len(self._indexdata)
                  def __iter__(self):
                      return iter(range(len(self)))
                  def revs(self, start=0, stop=None):
                      step = 1
                      if stop is not None:
                          if start > stop:
                              step = -1
                          stop += step
                      else:
                          stop = len(self)
                      return range(start, stop, step)
                  def parents(self, node):
                      validatenode(node)
                      if node not in self._indexbynode:
                          raise KeyError('unknown node')
                      entry = self._indexbynode[node]
                      return entry[b'p1'], entry[b'p2']
                  def parentrevs(self, rev):
                      p1, p2 = self.parents(self._indexbyrev[rev][b'node'])
                      return self.rev(p1), self.rev(p2)
                  def rev(self, node):
                      validatenode(node)
                      try:
                          self._indexbynode[node]
                      except KeyError:
                          raise error.LookupError(node, self._indexpath, _('no node'))
                      for rev, entry in self._indexbyrev.items():
                          if entry[b'node'] == node:
                              return rev
                      raise error.ProgrammingError(b'this should not occur')
                  def node(self, rev):
                      validaterev(rev)
                      return self._indexbyrev[rev][b'node']
                  def hasnode(self, node):
                      validatenode(node)
                      return node in self._indexbynode
                  def censorrevision(self, tr, censornode, tombstone=b''):
                      raise NotImplementedError('TODO')
                  def lookup(self, node):
                      if isinstance(node, int):
                          return self.node(node)
                      if len(node) == 20:
                          self.rev(node)
                          return node
                      try:
                          rev = int(node)
                          if '%d' % rev != node:
                              raise ValueError
                          if rev < 0:
                              rev = len(self) + rev
                          if rev < 0 or rev >= len(self):
                              raise ValueError
                          return self.node(rev)
                      except (ValueError, OverflowError):
                          pass
                      if len(node) == 40:
                          try:
                              rawnode = bin(node)
                              self.rev(rawnode)
                              return rawnode
                          except TypeError:
                              pass
                      raise error.LookupError(node, self._path, _('invalid lookup input'))
                  def linkrev(self, rev):
                      validaterev(rev)
                      return self._indexbyrev[rev][b'linkrev']
                  def _flags(self, rev):
                      validaterev(rev)
                      return self._indexbyrev[rev][b'flags']
                  def _candelta(self, baserev, rev):
                      validaterev(baserev)
                      validaterev(rev)
                      if (self._flags(baserev) & revlog.REVIDX_RAWTEXT_CHANGING_FLAGS) or (
                          self._flags(rev) & revlog.REVIDX_RAWTEXT_CHANGING_FLAGS
                      ):
                          return False
                      return True
                  def checkhash(self, text, node, p1=None, p2=None, rev=None):
                      if p1 is None and p2 is None:
                          p1, p2 = self.parents(node)
                      if node != storageutil.hashrevisionsha1(text, p1, p2):
                          raise simplestoreerror(
                              _("integrity check failed on %s") % self._path
                          )
                  def revision(self, nodeorrev, raw=False):
                      if isinstance(nodeorrev, int):
                          node = self.node(nodeorrev)
                      else:
                          node = nodeorrev
                      validatenode(node)
                      if node == self._repo.nullid:
                          return b''
                      rev = self.rev(node)
                      flags = self._flags(rev)
                      path = b'/'.join([self._storepath, hex(node)])
                      rawtext = self._svfs.read(path)
                      if raw:
                          validatehash = flagutil.processflagsraw(self, rawtext, flags)
                          text = rawtext
                      else:
                          r = flagutil.processflagsread(self, rawtext, flags)
                          text, validatehash = r
                      if validatehash:
                          self.checkhash(text, node, rev=rev)
                      return text
                  def rawdata(self, nodeorrev):
                      return self.revision(raw=True)
                  def read(self, node):
                      validatenode(node)
                      revision = self.revision(node)
                      if not revision.startswith(b'\1\n'):
                          return revision
                      start = revision.index(b'\1\n', 2)
                      return revision[start + 2 :]
                  def renamed(self, node):
                      validatenode(node)
                      if self.parents(node)[0] != self._repo.nullid:
                          return False
                      fulltext = self.revision(node)
                      m = storageutil.parsemeta(fulltext)[0]
                      if m and 'copy' in m:
                          return m['copy'], bin(m['copyrev'])
                      return False
                  def cmp(self, node, text):
                      validatenode(node)
                      t = text
                      if text.startswith(b'\1\n'):
                          t = b'\1\n\1\n' + text
                      p1, p2 = self.parents(node)
                      if storageutil.hashrevisionsha1(t, p1, p2) == node:
                          return False
                      if self.iscensored(self.rev(node)):
                          return text != b''
                      if self.renamed(node):
                          t2 = self.read(node)
                          return t2 != text
                      return True
                  def size(self, rev):
                      validaterev(rev)
                      node = self._indexbyrev[rev][b'node']
                      if self.renamed(node):
                          return len(self.read(node))
                      if self.iscensored(rev):
                          return 0
                      return len(self.revision(node))
                  def iscensored(self, rev):
                      validaterev(rev)
                      return self._flags(rev) & repository.REVISION_FLAG_CENSORED
                  def commonancestorsheads(self, a, b):
                      validatenode(a)
                      validatenode(b)
                      a = self.rev(a)
                      b = self.rev(b)
                      ancestors = ancestor.commonancestorsheads(self.parentrevs, a, b)
                      return pycompat.maplist(self.node, ancestors)
                  def descendants(self, revs):
                      # This is a copy of revlog.descendants()
                      first = min(revs)
                      if first == nullrev:
                          for i in self:
                              yield i
                          return
                      seen = set(revs)
                      for i in self.revs(start=first + 1):
                          for x in self.parentrevs(i):
                              if x != nullrev and x in seen:
                                  seen.add(i)
                                  yield i
                                  break
                  # Required by verify.
                  def files(self):
                      entries = self._svfs.listdir(self._storepath)
                      # Strip out undo.backup.* files created as part of transaction
                      # recording.
                      entries = [f for f in entries if not f.startswith('undo.backup.')]
                      return [b'/'.join((self._storepath, f)) for f in entries]
                  def storageinfo(
                      self,
                      exclusivefiles=False,
                      sharedfiles=False,
                      revisionscount=False,
                      trackedsize=False,
                      storedsize=False,
                  ):
                      # TODO do a real implementation of this
                      return {
                          'exclusivefiles': [],
                          'sharedfiles': [],
                          'revisionscount': len(self),
                          'trackedsize': 0,
                          'storedsize': None,
                      }
                  def verifyintegrity(self, state):
                      state['skipread'] = set()
                      for rev in self:
                          node = self.node(rev)
                          try:
                              self.revision(node)
                          except Exception as e:
                              yield simplefilestoreproblem(
                                  error='unpacking %s: %s' % (node, e), node=node
                              )
                              state['skipread'].add(node)
                  def emitrevisions(
                      self,
                      nodes,
                      nodesorder=None,
                      revisiondata=False,
                      assumehaveparentrevisions=False,
                      deltamode=repository.CG_DELTAMODE_STD,
                      sidedata_helpers=None,
                  ):
                      # TODO this will probably break on some ordering options.
                      nodes = [n for n in nodes if n != self._repo.nullid]
                      if not nodes:
                          return
                      for delta in storageutil.emitrevisions(
                          self,
                          nodes,
                          nodesorder,
                          simplestorerevisiondelta,
                          revisiondata=revisiondata,
                          assumehaveparentrevisions=assumehaveparentrevisions,
                          deltamode=deltamode,
                          sidedata_helpers=sidedata_helpers,
                      ):
                          yield delta
                  def add(self, text, meta, transaction, linkrev, p1, p2):
                      if meta or text.startswith(b'\1\n'):
                          text = storageutil.packmeta(meta, text)
                      return self.addrevision(text, transaction, linkrev, p1, p2)
                  def addrevision(
                      self,
                      text,
                      transaction,
                      linkrev,
                      p1,
                      p2,
                      node=None,
                      flags=revlog.REVIDX_DEFAULT_FLAGS,
                      cachedelta=None,
                  ):
                      validatenode(p1)
                      validatenode(p2)
                      if flags:
                          node = node or storageutil.hashrevisionsha1(text, p1, p2)
                      rawtext, validatehash = flagutil.processflagswrite(self, text, flags)
                      node = node or storageutil.hashrevisionsha1(text, p1, p2)
                      if node in self._indexbynode:
                          return node
                      if validatehash:
                          self.checkhash(rawtext, node, p1=p1, p2=p2)
                      return self._addrawrevision(
                          node, rawtext, transaction, linkrev, p1, p2, flags
                      )
                  def _addrawrevision(self, node, rawtext, transaction, link, p1, p2, flags):
                      transaction.addbackup(self._indexpath)
                      path = b'/'.join([self._storepath, hex(node)])
                      self._svfs.write(path, rawtext)
                      self._indexdata.append(
                          {
                              b'node': node,
                              b'p1': p1,
                              b'p2': p2,
                              b'linkrev': link,
                              b'flags': flags,
                          }
                      )
                      self._reflectindexupdate()
                      return node
                  def _reflectindexupdate(self):
                      self._refreshindex()
                      self._svfs.write(
                          self._indexpath, ''.join(cborutil.streamencode(self._indexdata))
                      )
                  def addgroup(
                      self,
                      deltas,
                      linkmapper,
                      transaction,
                      addrevisioncb=None,
                      duplicaterevisioncb=None,
                      maybemissingparents=False,
                  ):
                      if maybemissingparents:
                          raise error.Abort(
                              _('simple store does not support missing parents ' 'write mode')
                          )
                      empty = True
                      transaction.addbackup(self._indexpath)
                      for node, p1, p2, linknode, deltabase, delta, flags in deltas:
                          linkrev = linkmapper(linknode)
                          flags = flags or revlog.REVIDX_DEFAULT_FLAGS
                          if node in self._indexbynode:
                              if duplicaterevisioncb:
                                  duplicaterevisioncb(self, self.rev(node))
                              empty = False
                              continue
                          # Need to resolve the fulltext from the delta base.
                          if deltabase == self._repo.nullid:
                              text = mdiff.patch(b'', delta)
                          else:
                              text = mdiff.patch(self.revision(deltabase), delta)
                          rev = self._addrawrevision(
                              node, text, transaction, linkrev, p1, p2, flags
                          )
                          if addrevisioncb:
                              addrevisioncb(self, rev)
                          empty = False
                      return not empty
                  def _headrevs(self):
                      # Assume all revisions are heads by default.
                      revishead = {rev: True for rev in self._indexbyrev}
                      for rev, entry in self._indexbyrev.items():
                          # Unset head flag for all seen parents.
                          revishead[self.rev(entry[b'p1'])] = False
                          revishead[self.rev(entry[b'p2'])] = False
                      return [rev for rev, ishead in sorted(revishead.items()) if ishead]
                  def heads(self, start=None, stop=None):
                      # This is copied from revlog.py.
                      if start is None and stop is None:
                          if not len(self):
                              return [self._repo.nullid]
                          return [self.node(r) for r in self._headrevs()]
                      if start is None:
                          start = self._repo.nullid
                      if stop is None:
                          stop = []
                      stoprevs = {self.rev(n) for n in stop}
                      startrev = self.rev(start)
                      reachable = {startrev}
                      heads = {startrev}
                      parentrevs = self.parentrevs
                      for r in self.revs(start=startrev + 1):
                          for p in parentrevs(r):
                              if p in reachable:
                                  if r not in stoprevs:
                                      reachable.add(r)
                                  heads.add(r)
                              if p in heads and p not in stoprevs:
                                  heads.remove(p)
                      return [self.node(r) for r in heads]
                  def children(self, node):
                      validatenode(node)
                      # This is a copy of revlog.children().
                      c = []
                      p = self.rev(node)
                      for r in self.revs(start=p + 1):
                          prevs = [pr for pr in self.parentrevs(r) if pr != nullrev]
                          if prevs:
                              for pr in prevs:
                                  if pr == p:
                                      c.append(self.node(r))
                          elif p == nullrev:
                              c.append(self.node(r))
                      return c
                  def getstrippoint(self, minlink):
                      return storageutil.resolvestripinfo(
                          minlink,
                          len(self) - 1,
                          self._headrevs(),
                          self.linkrev,
                          self.parentrevs,
                      )
                  def strip(self, minlink, transaction):
                      if not len(self):
                          return
                      rev, _ignored = self.getstrippoint(minlink)
                      if rev == len(self):
                          return
                      # Purge index data starting at the requested revision.
                      self._indexdata[rev:] = []
                      self._reflectindexupdate()
              def issimplestorefile(f, kind, st):
                  if kind != stat.S_IFREG:
                      return False
                  if store.isrevlog(f, kind, st):
                      return False
                  # Ignore transaction undo files.
                  if f.startswith('undo.'):
                      return False
                  # Otherwise assume it belongs to the simple store.
                  return True
              class simplestore(store.encodedstore):
-                 def datafiles(self, undecodable=None):
-                     for x in super(simplestore, self).datafiles():
+                 def data_entries(self, undecodable=None):
+                     for x in super(simplestore, self).data_entries():
                          yield x
                      # Supplement with non-revlog files.
                      extrafiles = self._walk('data', True, filefilter=issimplestorefile)
                      for f1, size in extrafiles:
                          try:
                              f2 = store.decodefilename(f1)
                          except KeyError:
                              if undecodable is None:
                                  raise error.StorageError(b'undecodable revlog name %s' % f1)
                              else:
                                  undecodable.append(f1)
                                  continue
                          yield f2, size
              def reposetup(ui, repo):
                  if not repo.local():
                      return
                  if isinstance(repo, bundlerepo.bundlerepository):
                      raise error.Abort(_('cannot use simple store with bundlerepo'))
                  class simplestorerepo(repo.__class__):
                      def file(self, f):
                          return filestorage(repo, self.svfs, f)
                  repo.__class__ = simplestorerepo
              def featuresetup(ui, supported):
                  supported.add(REQUIREMENT)
              def newreporequirements(orig, ui, createopts):
                  """Modifies default requirements for new repos to use the simple store."""
                  requirements = orig(ui, createopts)
                  # These requirements are only used to affect creation of the store
                  # object. We have our own store. So we can remove them.
                  # TODO do this once we feel like taking the test hit.
                  # if 'fncache' in requirements:
                  #    requirements.remove('fncache')
                  # if 'dotencode' in requirements:
                  #    requirements.remove('dotencode')
                  requirements.add(REQUIREMENT)
                  return requirements
              def makestore(orig, requirements, path, vfstype):
                  if REQUIREMENT not in requirements:
                      return orig(requirements, path, vfstype)
                  return simplestore(path, vfstype)
              def verifierinit(orig, self, *args, **kwargs):
                  orig(self, *args, **kwargs)
                  # We don't care that files in the store don't align with what is
                  # advertised. So suppress these warnings.
                  self.warnorphanstorefiles = False
              def extsetup(ui):
                  localrepo.featuresetupfuncs.add(featuresetup)
                  extensions.wrapfunction(
                      localrepo, 'newreporequirements', newreporequirements
                  )
                  extensions.wrapfunction(localrepo, 'makestore', makestore)
                  extensions.wrapfunction(verify.verifier, '__init__', verifierinit)

General Comments 0

Write
Preview

You need to be logged in to leave comments. Login now

No reviewers

No TODOs yet

	Site-wide shortcuts
/	Use quick search box
g h	Goto home page
g g	Goto my private gists page
g G	Goto my public gists page
g 0-9	Goto bookmarked items from 0-9
n r	New repository page
n g	New gist page

	Repositories
g s	Goto summary page
g c	Goto changelog page
g f	Goto files page
g F	Goto files page with file search activated
g p	Goto pull requests page
g o	Goto repository settings
g O	Goto repository access permissions settings
t s	Toggle sidebar on some pages