upstream/mercurial-mirror Commit - r47657:6085b7f1

store: also return some information about the type of file `walk` found...

marmoute -

r47657:6085b7f1 default

parent child

hgext/largefiles/lfutil.py

0 +1 -1

             # Copyright 2009-2010 Gregory P. Ward
             # Copyright 2009-2010 Intelerad Medical Systems Incorporated
             # Copyright 2010-2011 Fog Creek Software
             # Copyright 2010-2011 Unity Technologies
             #
             # This software may be used and distributed according to the terms of the
             # GNU General Public License version 2 or any later version.
             '''largefiles utility code: must not import other modules in this package.'''
             from __future__ import absolute_import
             import contextlib
             import copy
             import os
             import stat
             from mercurial.i18n import _
             from mercurial.node import (
                 hex,
                 nullid,
             )
             from mercurial.pycompat import open
             from mercurial import (
                 dirstate,
                 encoding,
                 error,
                 httpconnection,
                 match as matchmod,
                 pycompat,
                 scmutil,
                 sparse,
                 util,
                 vfs as vfsmod,
             )
             from mercurial.utils import hashutil
             shortname = b'.hglf'
             shortnameslash = shortname + b'/'
             longname = b'largefiles'
             # -- Private worker functions ------------------------------------------
             @contextlib.contextmanager
             def lfstatus(repo, value=True):
                 oldvalue = getattr(repo, 'lfstatus', False)
                 repo.lfstatus = value
                 try:
                     yield
                 finally:
                     repo.lfstatus = oldvalue
             def getminsize(ui, assumelfiles, opt, default=10):
                 lfsize = opt
                 if not lfsize and assumelfiles:
                     lfsize = ui.config(longname, b'minsize', default=default)
                 if lfsize:
                     try:
                         lfsize = float(lfsize)
                     except ValueError:
                         raise error.Abort(
                             _(b'largefiles: size must be number (not %s)\n') % lfsize
                         )
                 if lfsize is None:
                     raise error.Abort(_(b'minimum size for largefiles must be specified'))
                 return lfsize
             def link(src, dest):
                 """Try to create hardlink - if that fails, efficiently make a copy."""
                 util.makedirs(os.path.dirname(dest))
                 try:
                     util.oslink(src, dest)
                 except OSError:
                     # if hardlinks fail, fallback on atomic copy
                     with open(src, b'rb') as srcf, util.atomictempfile(dest) as dstf:
                         for chunk in util.filechunkiter(srcf):
                             dstf.write(chunk)
                     os.chmod(dest, os.stat(src).st_mode)
             def usercachepath(ui, hash):
                 """Return the correct location in the "global" largefiles cache for a file
                 with the given hash.
                 This cache is used for sharing of largefiles across repositories - both
                 to preserve download bandwidth and storage space."""
                 return os.path.join(_usercachedir(ui), hash)
             def _usercachedir(ui, name=longname):
                 '''Return the location of the "global" largefiles cache.'''
                 path = ui.configpath(name, b'usercache')
                 if path:
                     return path
                 hint = None
                 if pycompat.iswindows:
                     appdata = encoding.environ.get(
                         b'LOCALAPPDATA', encoding.environ.get(b'APPDATA')
                     )
                     if appdata:
                         return os.path.join(appdata, name)
                     hint = _(b"define %s or %s in the environment, or set %s.usercache") % (
                         b"LOCALAPPDATA",
                         b"APPDATA",
                         name,
                     )
                 elif pycompat.isdarwin:
                     home = encoding.environ.get(b'HOME')
                     if home:
                         return os.path.join(home, b'Library', b'Caches', name)
                     hint = _(b"define %s in the environment, or set %s.usercache") % (
                         b"HOME",
                         name,
                     )
                 elif pycompat.isposix:
                     path = encoding.environ.get(b'XDG_CACHE_HOME')
                     if path:
                         return os.path.join(path, name)
                     home = encoding.environ.get(b'HOME')
                     if home:
                         return os.path.join(home, b'.cache', name)
                     hint = _(b"define %s or %s in the environment, or set %s.usercache") % (
                         b"XDG_CACHE_HOME",
                         b"HOME",
                         name,
                     )
                 else:
                     raise error.Abort(
                         _(b'unknown operating system: %s\n') % pycompat.osname
                     )
                 raise error.Abort(_(b'unknown %s usercache location') % name, hint=hint)
             def inusercache(ui, hash):
                 path = usercachepath(ui, hash)
                 return os.path.exists(path)
             def findfile(repo, hash):
                 """Return store path of the largefile with the specified hash.
                 As a side effect, the file might be linked from user cache.
                 Return None if the file can't be found locally."""
                 path, exists = findstorepath(repo, hash)
                 if exists:
                     repo.ui.note(_(b'found %s in store\n') % hash)
                     return path
                 elif inusercache(repo.ui, hash):
                     repo.ui.note(_(b'found %s in system cache\n') % hash)
                     path = storepath(repo, hash)
                     link(usercachepath(repo.ui, hash), path)
                     return path
                 return None
             class largefilesdirstate(dirstate.dirstate):
                 def __getitem__(self, key):
                     return super(largefilesdirstate, self).__getitem__(unixpath(key))
                 def normal(self, f):
                     return super(largefilesdirstate, self).normal(unixpath(f))
                 def remove(self, f):
                     return super(largefilesdirstate, self).remove(unixpath(f))
                 def add(self, f):
                     return super(largefilesdirstate, self).add(unixpath(f))
                 def drop(self, f):
                     return super(largefilesdirstate, self).drop(unixpath(f))
                 def forget(self, f):
                     return super(largefilesdirstate, self).forget(unixpath(f))
                 def normallookup(self, f):
                     return super(largefilesdirstate, self).normallookup(unixpath(f))
                 def _ignore(self, f):
                     return False
                 def write(self, tr=False):
                     # (1) disable PENDING mode always
                     #     (lfdirstate isn't yet managed as a part of the transaction)
                     # (2) avoid develwarn 'use dirstate.write with ....'
                     super(largefilesdirstate, self).write(None)
             def openlfdirstate(ui, repo, create=True):
                 """
                 Return a dirstate object that tracks largefiles: i.e. its root is
                 the repo root, but it is saved in .hg/largefiles/dirstate.
                 """
                 vfs = repo.vfs
                 lfstoredir = longname
                 opener = vfsmod.vfs(vfs.join(lfstoredir))
                 lfdirstate = largefilesdirstate(
                     opener,
                     ui,
                     repo.root,
                     repo.dirstate._validate,
                     lambda: sparse.matcher(repo),
                     repo.nodeconstants,
                 )
                 # If the largefiles dirstate does not exist, populate and create
                 # it. This ensures that we create it on the first meaningful
                 # largefiles operation in a new clone.
                 if create and not vfs.exists(vfs.join(lfstoredir, b'dirstate')):
                     matcher = getstandinmatcher(repo)
                     standins = repo.dirstate.walk(
                         matcher, subrepos=[], unknown=False, ignored=False
                     )
                     if len(standins) > 0:
                         vfs.makedirs(lfstoredir)
                     for standin in standins:
                         lfile = splitstandin(standin)
                         lfdirstate.normallookup(lfile)
                 return lfdirstate
             def lfdirstatestatus(lfdirstate, repo):
                 pctx = repo[b'.']
                 match = matchmod.always()
                 unsure, s = lfdirstate.status(
                     match, subrepos=[], ignored=False, clean=False, unknown=False
                 )
                 modified, clean = s.modified, s.clean
                 for lfile in unsure:
                     try:
                         fctx = pctx[standin(lfile)]
                     except LookupError:
                         fctx = None
                     if not fctx or readasstandin(fctx) != hashfile(repo.wjoin(lfile)):
                         modified.append(lfile)
                     else:
                         clean.append(lfile)
                         lfdirstate.normal(lfile)
                 return s
             def listlfiles(repo, rev=None, matcher=None):
                 """return a list of largefiles in the working copy or the
                 specified changeset"""
                 if matcher is None:
                     matcher = getstandinmatcher(repo)
                 # ignore unknown files in working directory
                 return [
                     splitstandin(f)
                     for f in repo[rev].walk(matcher)
                     if rev is not None or repo.dirstate[f] != b'?'
                 ]
             def instore(repo, hash, forcelocal=False):
                 '''Return true if a largefile with the given hash exists in the store'''
                 return os.path.exists(storepath(repo, hash, forcelocal))
             def storepath(repo, hash, forcelocal=False):
                 """Return the correct location in the repository largefiles store for a
                 file with the given hash."""
                 if not forcelocal and repo.shared():
                     return repo.vfs.reljoin(repo.sharedpath, longname, hash)
                 return repo.vfs.join(longname, hash)
             def findstorepath(repo, hash):
                 """Search through the local store path(s) to find the file for the given
                 hash.  If the file is not found, its path in the primary store is returned.
                 The return value is a tuple of (path, exists(path)).
                 """
                 # For shared repos, the primary store is in the share source.  But for
                 # backward compatibility, force a lookup in the local store if it wasn't
                 # found in the share source.
                 path = storepath(repo, hash, False)
                 if instore(repo, hash):
                     return (path, True)
                 elif repo.shared() and instore(repo, hash, True):
                     return storepath(repo, hash, True), True
                 return (path, False)
             def copyfromcache(repo, hash, filename):
                 """Copy the specified largefile from the repo or system cache to
                 filename in the repository. Return true on success or false if the
                 file was not found in either cache (which should not happened:
                 this is meant to be called only after ensuring that the needed
                 largefile exists in the cache)."""
                 wvfs = repo.wvfs
                 path = findfile(repo, hash)
                 if path is None:
                     return False
                 wvfs.makedirs(wvfs.dirname(wvfs.join(filename)))
                 # The write may fail before the file is fully written, but we
                 # don't use atomic writes in the working copy.
                 with open(path, b'rb') as srcfd, wvfs(filename, b'wb') as destfd:
                     gothash = copyandhash(util.filechunkiter(srcfd), destfd)
                 if gothash != hash:
                     repo.ui.warn(
                         _(b'%s: data corruption in %s with hash %s\n')
                         % (filename, path, gothash)
                     )
                     wvfs.unlink(filename)
                     return False
                 return True
             def copytostore(repo, ctx, file, fstandin):
                 wvfs = repo.wvfs
                 hash = readasstandin(ctx[fstandin])
                 if instore(repo, hash):
                     return
                 if wvfs.exists(file):
                     copytostoreabsolute(repo, wvfs.join(file), hash)
                 else:
                     repo.ui.warn(
                         _(b"%s: largefile %s not available from local store\n")
                         % (file, hash)
                     )
             def copyalltostore(repo, node):
                 '''Copy all largefiles in a given revision to the store'''
                 ctx = repo[node]
                 for filename in ctx.files():
                     realfile = splitstandin(filename)
                     if realfile is not None and filename in ctx.manifest():
                         copytostore(repo, ctx, realfile, filename)
             def copytostoreabsolute(repo, file, hash):
                 if inusercache(repo.ui, hash):
                     link(usercachepath(repo.ui, hash), storepath(repo, hash))
                 else:
                     util.makedirs(os.path.dirname(storepath(repo, hash)))
                     with open(file, b'rb') as srcf:
                         with util.atomictempfile(
                             storepath(repo, hash), createmode=repo.store.createmode
                         ) as dstf:
                             for chunk in util.filechunkiter(srcf):
                                 dstf.write(chunk)
                     linktousercache(repo, hash)
             def linktousercache(repo, hash):
                 """Link / copy the largefile with the specified hash from the store
                 to the cache."""
                 path = usercachepath(repo.ui, hash)
                 link(storepath(repo, hash), path)
             def getstandinmatcher(repo, rmatcher=None):
                 '''Return a match object that applies rmatcher to the standin directory'''
                 wvfs = repo.wvfs
                 standindir = shortname
                 # no warnings about missing files or directories
                 badfn = lambda f, msg: None
                 if rmatcher and not rmatcher.always():
                     pats = [wvfs.join(standindir, pat) for pat in rmatcher.files()]
                     if not pats:
                         pats = [wvfs.join(standindir)]
                     match = scmutil.match(repo[None], pats, badfn=badfn)
                 else:
                     # no patterns: relative to repo root
                     match = scmutil.match(repo[None], [wvfs.join(standindir)], badfn=badfn)
                 return match
             def composestandinmatcher(repo, rmatcher):
                 """Return a matcher that accepts standins corresponding to the
                 files accepted by rmatcher. Pass the list of files in the matcher
                 as the paths specified by the user."""
                 smatcher = getstandinmatcher(repo, rmatcher)
                 isstandin = smatcher.matchfn
                 def composedmatchfn(f):
                     return isstandin(f) and rmatcher.matchfn(splitstandin(f))
                 smatcher.matchfn = composedmatchfn
                 return smatcher
             def standin(filename):
                 """Return the repo-relative path to the standin for the specified big
                 file."""
                 # Notes:
                 # 1) Some callers want an absolute path, but for instance addlargefiles
                 #    needs it repo-relative so it can be passed to repo[None].add().  So
                 #    leave it up to the caller to use repo.wjoin() to get an absolute path.
                 # 2) Join with '/' because that's what dirstate always uses, even on
                 #    Windows. Change existing separator to '/' first in case we are
                 #    passed filenames from an external source (like the command line).
                 return shortnameslash + util.pconvert(filename)
             def isstandin(filename):
                 """Return true if filename is a big file standin. filename must be
                 in Mercurial's internal form (slash-separated)."""
                 return filename.startswith(shortnameslash)
             def splitstandin(filename):
                 # Split on / because that's what dirstate always uses, even on Windows.
                 # Change local separator to / first just in case we are passed filenames
                 # from an external source (like the command line).
                 bits = util.pconvert(filename).split(b'/', 1)
                 if len(bits) == 2 and bits[0] == shortname:
                     return bits[1]
                 else:
                     return None
             def updatestandin(repo, lfile, standin):
                 """Re-calculate hash value of lfile and write it into standin
                 This assumes that "lfutil.standin(lfile) == standin", for efficiency.
                 """
                 file = repo.wjoin(lfile)
                 if repo.wvfs.exists(lfile):
                     hash = hashfile(file)
                     executable = getexecutable(file)
                     writestandin(repo, standin, hash, executable)
                 else:
                     raise error.Abort(_(b'%s: file not found!') % lfile)
             def readasstandin(fctx):
                 """read hex hash from given filectx of standin file
                 This encapsulates how "standin" data is stored into storage layer."""
                 return fctx.data().strip()
             def writestandin(repo, standin, hash, executable):
                 '''write hash to <repo.root>/<standin>'''
                 repo.wwrite(standin, hash + b'\n', executable and b'x' or b'')
             def copyandhash(instream, outfile):
                 """Read bytes from instream (iterable) and write them to outfile,
                 computing the SHA-1 hash of the data along the way. Return the hash."""
                 hasher = hashutil.sha1(b'')
                 for data in instream:
                     hasher.update(data)
                     outfile.write(data)
                 return hex(hasher.digest())
             def hashfile(file):
                 if not os.path.exists(file):
                     return b''
                 with open(file, b'rb') as fd:
                     return hexsha1(fd)
             def getexecutable(filename):
                 mode = os.stat(filename).st_mode
                 return (
                     (mode & stat.S_IXUSR)
                     and (mode & stat.S_IXGRP)
                     and (mode & stat.S_IXOTH)
                 )
             def urljoin(first, second, *arg):
                 def join(left, right):
                     if not left.endswith(b'/'):
                         left += b'/'
                     if right.startswith(b'/'):
                         right = right[1:]
                     return left + right
                 url = join(first, second)
                 for a in arg:
                     url = join(url, a)
                 return url
             def hexsha1(fileobj):
                 """hexsha1 returns the hex-encoded sha1 sum of the data in the file-like
                 object data"""
                 h = hashutil.sha1()
                 for chunk in util.filechunkiter(fileobj):
                     h.update(chunk)
                 return hex(h.digest())
             def httpsendfile(ui, filename):
                 return httpconnection.httpsendfile(ui, filename, b'rb')
             def unixpath(path):
                 '''Return a version of path normalized for use with the lfdirstate.'''
                 return util.pconvert(os.path.normpath(path))
             def islfilesrepo(repo):
                 '''Return true if the repo is a largefile repo.'''
                 if b'largefiles' in repo.requirements and any(
-                    shortnameslash in f[0] for f in repo.store.datafiles()
+                    shortnameslash in f[1] for f in repo.store.datafiles()
                 ):
                     return True
                 return any(openlfdirstate(repo.ui, repo, False))
             class storeprotonotcapable(Exception):
                 def __init__(self, storetypes):
                     self.storetypes = storetypes
             def getstandinsstate(repo):
                 standins = []
                 matcher = getstandinmatcher(repo)
                 wctx = repo[None]
                 for standin in repo.dirstate.walk(
                     matcher, subrepos=[], unknown=False, ignored=False
                 ):
                     lfile = splitstandin(standin)
                     try:
                         hash = readasstandin(wctx[standin])
                     except IOError:
                         hash = None
                     standins.append((lfile, hash))
                 return standins
             def synclfdirstate(repo, lfdirstate, lfile, normallookup):
                 lfstandin = standin(lfile)
                 if lfstandin in repo.dirstate:
                     stat = repo.dirstate._map[lfstandin]
                     state, mtime = stat[0], stat[3]
                 else:
                     state, mtime = b'?', -1
                 if state == b'n':
                     if normallookup or mtime < 0 or not repo.wvfs.exists(lfile):
                         # state 'n' doesn't ensure 'clean' in this case
                         lfdirstate.normallookup(lfile)
                     else:
                         lfdirstate.normal(lfile)
                 elif state == b'm':
                     lfdirstate.normallookup(lfile)
                 elif state == b'r':
                     lfdirstate.remove(lfile)
                 elif state == b'a':
                     lfdirstate.add(lfile)
                 elif state == b'?':
                     lfdirstate.drop(lfile)
             def markcommitted(orig, ctx, node):
                 repo = ctx.repo()
                 orig(node)
                 # ATTENTION: "ctx.files()" may differ from "repo[node].files()"
                 # because files coming from the 2nd parent are omitted in the latter.
                 #
                 # The former should be used to get targets of "synclfdirstate",
                 # because such files:
                 # - are marked as "a" by "patch.patch()" (e.g. via transplant), and
                 # - have to be marked as "n" after commit, but
                 # - aren't listed in "repo[node].files()"
                 lfdirstate = openlfdirstate(repo.ui, repo)
                 for f in ctx.files():
                     lfile = splitstandin(f)
                     if lfile is not None:
                         synclfdirstate(repo, lfdirstate, lfile, False)
                 lfdirstate.write()
                 # As part of committing, copy all of the largefiles into the cache.
                 #
                 # Using "node" instead of "ctx" implies additional "repo[node]"
                 # lookup while copyalltostore(), but can omit redundant check for
                 # files comming from the 2nd parent, which should exist in store
                 # at merging.
                 copyalltostore(repo, node)
             def getlfilestoupdate(oldstandins, newstandins):
                 changedstandins = set(oldstandins).symmetric_difference(set(newstandins))
                 filelist = []
                 for f in changedstandins:
                     if f[0] not in filelist:
                         filelist.append(f[0])
                 return filelist
             def getlfilestoupload(repo, missing, addfunc):
                 makeprogress = repo.ui.makeprogress
                 with makeprogress(
                     _(b'finding outgoing largefiles'),
                     unit=_(b'revisions'),
                     total=len(missing),
                 ) as progress:
                     for i, n in enumerate(missing):
                         progress.update(i)
                         parents = [p for p in repo[n].parents() if p != nullid]
                         with lfstatus(repo, value=False):
                             ctx = repo[n]
                         files = set(ctx.files())
                         if len(parents) == 2:
                             mc = ctx.manifest()
                             mp1 = ctx.p1().manifest()
                             mp2 = ctx.p2().manifest()
                             for f in mp1:
                                 if f not in mc:
                                     files.add(f)
                             for f in mp2:
                                 if f not in mc:
                                     files.add(f)
                             for f in mc:
                                 if mc[f] != mp1.get(f, None) or mc[f] != mp2.get(f, None):
                                     files.add(f)
                         for fn in files:
                             if isstandin(fn) and fn in ctx:
                                 addfunc(fn, readasstandin(ctx[fn]))
             def updatestandinsbymatch(repo, match):
                 """Update standins in the working directory according to specified match
                 This returns (possibly modified) ``match`` object to be used for
                 subsequent commit process.
                 """
                 ui = repo.ui
                 # Case 1: user calls commit with no specific files or
                 # include/exclude patterns: refresh and commit all files that
                 # are "dirty".
                 if match is None or match.always():
                     # Spend a bit of time here to get a list of files we know
                     # are modified so we can compare only against those.
                     # It can cost a lot of time (several seconds)
                     # otherwise to update all standins if the largefiles are
                     # large.
                     lfdirstate = openlfdirstate(ui, repo)
                     dirtymatch = matchmod.always()
                     unsure, s = lfdirstate.status(
                         dirtymatch, subrepos=[], ignored=False, clean=False, unknown=False
                     )
                     modifiedfiles = unsure + s.modified + s.added + s.removed
                     lfiles = listlfiles(repo)
                     # this only loops through largefiles that exist (not
                     # removed/renamed)
                     for lfile in lfiles:
                         if lfile in modifiedfiles:
                             fstandin = standin(lfile)
                             if repo.wvfs.exists(fstandin):
                                 # this handles the case where a rebase is being
                                 # performed and the working copy is not updated
                                 # yet.
                                 if repo.wvfs.exists(lfile):
                                     updatestandin(repo, lfile, fstandin)
                     return match
                 lfiles = listlfiles(repo)
                 match._files = repo._subdirlfs(match.files(), lfiles)
                 # Case 2: user calls commit with specified patterns: refresh
                 # any matching big files.
                 smatcher = composestandinmatcher(repo, match)
                 standins = repo.dirstate.walk(
                     smatcher, subrepos=[], unknown=False, ignored=False
                 )
                 # No matching big files: get out of the way and pass control to
                 # the usual commit() method.
                 if not standins:
                     return match
                 # Refresh all matching big files.  It's possible that the
                 # commit will end up failing, in which case the big files will
                 # stay refreshed.  No harm done: the user modified them and
                 # asked to commit them, so sooner or later we're going to
                 # refresh the standins.  Might as well leave them refreshed.
                 lfdirstate = openlfdirstate(ui, repo)
                 for fstandin in standins:
                     lfile = splitstandin(fstandin)
                     if lfdirstate[lfile] != b'r':
                         updatestandin(repo, lfile, fstandin)
                 # Cook up a new matcher that only matches regular files or
                 # standins corresponding to the big files requested by the
                 # user.  Have to modify _files to prevent commit() from
                 # complaining "not tracked" for big files.
                 match = copy.copy(match)
                 origmatchfn = match.matchfn
                 # Check both the list of largefiles and the list of
                 # standins because if a largefile was removed, it
                 # won't be in the list of largefiles at this point
                 match._files += sorted(standins)
                 actualfiles = []
                 for f in match._files:
                     fstandin = standin(f)
                     # For largefiles, only one of the normal and standin should be
                     # committed (except if one of them is a remove).  In the case of a
                     # standin removal, drop the normal file if it is unknown to dirstate.
                     # Thus, skip plain largefile names but keep the standin.
                     if f in lfiles or fstandin in standins:
                         if repo.dirstate[fstandin] != b'r':
                             if repo.dirstate[f] != b'r':
                                 continue
                         elif repo.dirstate[f] == b'?':
                             continue
                     actualfiles.append(f)
                 match._files = actualfiles
                 def matchfn(f):
                     if origmatchfn(f):
                         return f not in lfiles
                     else:
                         return f in standins
                 match.matchfn = matchfn
                 return match
             class automatedcommithook(object):
                 """Stateful hook to update standins at the 1st commit of resuming
                 For efficiency, updating standins in the working directory should
                 be avoided while automated committing (like rebase, transplant and
                 so on), because they should be updated before committing.
                 But the 1st commit of resuming automated committing (e.g. ``rebase
                 --continue``) should update them, because largefiles may be
                 modified manually.
                 """
                 def __init__(self, resuming):
                     self.resuming = resuming
                 def __call__(self, repo, match):
                     if self.resuming:
                         self.resuming = False  # avoids updating at subsequent commits
                         return updatestandinsbymatch(repo, match)
                     else:
                         return match
             def getstatuswriter(ui, repo, forcibly=None):
                 """Return the function to write largefiles specific status out
                 If ``forcibly`` is ``None``, this returns the last element of
                 ``repo._lfstatuswriters`` as "default" writer function.
                 Otherwise, this returns the function to always write out (or
                 ignore if ``not forcibly``) status.
                 """
                 if forcibly is None and util.safehasattr(repo, b'_largefilesenabled'):
                     return repo._lfstatuswriters[-1]
                 else:
                     if forcibly:
                         return ui.status  # forcibly WRITE OUT
                     else:
                         return lambda *msg, **opts: None  # forcibly IGNORE

hgext/largefiles/reposetup.py

0 +1 -1

             # Copyright 2009-2010 Gregory P. Ward
             # Copyright 2009-2010 Intelerad Medical Systems Incorporated
             # Copyright 2010-2011 Fog Creek Software
             # Copyright 2010-2011 Unity Technologies
             #
             # This software may be used and distributed according to the terms of the
             # GNU General Public License version 2 or any later version.
             '''setup for largefiles repositories: reposetup'''
             from __future__ import absolute_import
             import copy
             from mercurial.i18n import _
             from mercurial import (
                 error,
                 extensions,
                 localrepo,
                 match as matchmod,
                 scmutil,
                 util,
             )
             from . import (
                 lfcommands,
                 lfutil,
             )
             def reposetup(ui, repo):
                 # wire repositories should be given new wireproto functions
                 # by "proto.wirereposetup()" via "hg.wirepeersetupfuncs"
                 if not repo.local():
                     return
                 class lfilesrepo(repo.__class__):
                     # the mark to examine whether "repo" object enables largefiles or not
                     _largefilesenabled = True
                     lfstatus = False
                     # When lfstatus is set, return a context that gives the names
                     # of largefiles instead of their corresponding standins and
                     # identifies the largefiles as always binary, regardless of
                     # their actual contents.
                     def __getitem__(self, changeid):
                         ctx = super(lfilesrepo, self).__getitem__(changeid)
                         if self.lfstatus:
                             def files(orig):
                                 filenames = orig()
                                 return [lfutil.splitstandin(f) or f for f in filenames]
                             extensions.wrapfunction(ctx, 'files', files)
                             def manifest(orig):
                                 man1 = orig()
                                 class lfilesmanifest(man1.__class__):
                                     def __contains__(self, filename):
                                         orig = super(lfilesmanifest, self).__contains__
                                         return orig(filename) or orig(
                                             lfutil.standin(filename)
                                         )
                                 man1.__class__ = lfilesmanifest
                                 return man1
                             extensions.wrapfunction(ctx, 'manifest', manifest)
                             def filectx(orig, path, fileid=None, filelog=None):
                                 try:
                                     if filelog is not None:
                                         result = orig(path, fileid, filelog)
                                     else:
                                         result = orig(path, fileid)
                                 except error.LookupError:
                                     # Adding a null character will cause Mercurial to
                                     # identify this as a binary file.
                                     if filelog is not None:
                                         result = orig(lfutil.standin(path), fileid, filelog)
                                     else:
                                         result = orig(lfutil.standin(path), fileid)
                                     olddata = result.data
                                     result.data = lambda: olddata() + b'\0'
                                 return result
                             extensions.wrapfunction(ctx, 'filectx', filectx)
                         return ctx
                     # Figure out the status of big files and insert them into the
                     # appropriate list in the result. Also removes standin files
                     # from the listing. Revert to the original status if
                     # self.lfstatus is False.
                     # XXX large file status is buggy when used on repo proxy.
                     # XXX this needs to be investigated.
                     @localrepo.unfilteredmethod
                     def status(
                         self,
                         node1=b'.',
                         node2=None,
                         match=None,
                         ignored=False,
                         clean=False,
                         unknown=False,
                         listsubrepos=False,
                     ):
                         listignored, listclean, listunknown = ignored, clean, unknown
                         orig = super(lfilesrepo, self).status
                         if not self.lfstatus:
                             return orig(
                                 node1,
                                 node2,
                                 match,
                                 listignored,
                                 listclean,
                                 listunknown,
                                 listsubrepos,
                             )
                         # some calls in this function rely on the old version of status
                         self.lfstatus = False
                         ctx1 = self[node1]
                         ctx2 = self[node2]
                         working = ctx2.rev() is None
                         parentworking = working and ctx1 == self[b'.']
                         if match is None:
                             match = matchmod.always()
                         try:
                             # updating the dirstate is optional
                             # so we don't wait on the lock
                             wlock = self.wlock(False)
                             gotlock = True
                         except error.LockError:
                             wlock = util.nullcontextmanager()
                             gotlock = False
                         with wlock:
                             # First check if paths or patterns were specified on the
                             # command line.  If there were, and they don't match any
                             # largefiles, we should just bail here and let super
                             # handle it -- thus gaining a big performance boost.
                             lfdirstate = lfutil.openlfdirstate(ui, self)
                             if not match.always():
                                 for f in lfdirstate:
                                     if match(f):
                                         break
                                 else:
                                     return orig(
                                         node1,
                                         node2,
                                         match,
                                         listignored,
                                         listclean,
                                         listunknown,
                                         listsubrepos,
                                     )
                             # Create a copy of match that matches standins instead
                             # of largefiles.
                             def tostandins(files):
                                 if not working:
                                     return files
                                 newfiles = []
                                 dirstate = self.dirstate
                                 for f in files:
                                     sf = lfutil.standin(f)
                                     if sf in dirstate:
                                         newfiles.append(sf)
                                     elif dirstate.hasdir(sf):
                                         # Directory entries could be regular or
                                         # standin, check both
                                         newfiles.extend((f, sf))
                                     else:
                                         newfiles.append(f)
                                 return newfiles
                             m = copy.copy(match)
                             m._files = tostandins(m._files)
                             result = orig(
                                 node1, node2, m, ignored, clean, unknown, listsubrepos
                             )
                             if working:
                                 def sfindirstate(f):
                                     sf = lfutil.standin(f)
                                     dirstate = self.dirstate
                                     return sf in dirstate or dirstate.hasdir(sf)
                                 match._files = [f for f in match._files if sfindirstate(f)]
                                 # Don't waste time getting the ignored and unknown
                                 # files from lfdirstate
                                 unsure, s = lfdirstate.status(
                                     match,
                                     subrepos=[],
                                     ignored=False,
                                     clean=listclean,
                                     unknown=False,
                                 )
                                 (modified, added, removed, deleted, clean) = (
                                     s.modified,
                                     s.added,
                                     s.removed,
                                     s.deleted,
                                     s.clean,
                                 )
                                 if parentworking:
                                     for lfile in unsure:
                                         standin = lfutil.standin(lfile)
                                         if standin not in ctx1:
                                             # from second parent
                                             modified.append(lfile)
                                         elif lfutil.readasstandin(
                                             ctx1[standin]
                                         ) != lfutil.hashfile(self.wjoin(lfile)):
                                             modified.append(lfile)
                                         else:
                                             if listclean:
                                                 clean.append(lfile)
                                             lfdirstate.normal(lfile)
                                 else:
                                     tocheck = unsure + modified + added + clean
                                     modified, added, clean = [], [], []
                                     checkexec = self.dirstate._checkexec
                                     for lfile in tocheck:
                                         standin = lfutil.standin(lfile)
                                         if standin in ctx1:
                                             abslfile = self.wjoin(lfile)
                                             if (
                                                 lfutil.readasstandin(ctx1[standin])
                                                 != lfutil.hashfile(abslfile)
                                             ) or (
                                                 checkexec
                                                 and (b'x' in ctx1.flags(standin))
                                                 != bool(lfutil.getexecutable(abslfile))
                                             ):
                                                 modified.append(lfile)
                                             elif listclean:
                                                 clean.append(lfile)
                                         else:
                                             added.append(lfile)
                                     # at this point, 'removed' contains largefiles
                                     # marked as 'R' in the working context.
                                     # then, largefiles not managed also in the target
                                     # context should be excluded from 'removed'.
                                     removed = [
                                         lfile
                                         for lfile in removed
                                         if lfutil.standin(lfile) in ctx1
                                     ]
                                 # Standins no longer found in lfdirstate have been deleted
                                 for standin in ctx1.walk(lfutil.getstandinmatcher(self)):
                                     lfile = lfutil.splitstandin(standin)
                                     if not match(lfile):
                                         continue
                                     if lfile not in lfdirstate:
                                         deleted.append(lfile)
                                         # Sync "largefile has been removed" back to the
                                         # standin. Removing a file as a side effect of
                                         # running status is gross, but the alternatives (if
                                         # any) are worse.
                                         self.wvfs.unlinkpath(standin, ignoremissing=True)
                                 # Filter result lists
                                 result = list(result)
                                 # Largefiles are not really removed when they're
                                 # still in the normal dirstate. Likewise, normal
                                 # files are not really removed if they are still in
                                 # lfdirstate. This happens in merges where files
                                 # change type.
                                 removed = [f for f in removed if f not in self.dirstate]
                                 result[2] = [f for f in result[2] if f not in lfdirstate]
                                 lfiles = set(lfdirstate)
                                 # Unknown files
                                 result[4] = set(result[4]).difference(lfiles)
                                 # Ignored files
                                 result[5] = set(result[5]).difference(lfiles)
                                 # combine normal files and largefiles
                                 normals = [
                                     [fn for fn in filelist if not lfutil.isstandin(fn)]
                                     for filelist in result
                                 ]
                                 lfstatus = (
                                     modified,
                                     added,
                                     removed,
                                     deleted,
                                     [],
                                     [],
                                     clean,
                                 )
                                 result = [
                                     sorted(list1 + list2)
                                     for (list1, list2) in zip(normals, lfstatus)
                                 ]
                             else:  # not against working directory
                                 result = [
                                     [lfutil.splitstandin(f) or f for f in items]
                                     for items in result
                                 ]
                             if gotlock:
                                 lfdirstate.write()
                         self.lfstatus = True
                         return scmutil.status(*result)
                     def commitctx(self, ctx, *args, **kwargs):
                         node = super(lfilesrepo, self).commitctx(ctx, *args, **kwargs)
                         class lfilesctx(ctx.__class__):
                             def markcommitted(self, node):
                                 orig = super(lfilesctx, self).markcommitted
                                 return lfutil.markcommitted(orig, self, node)
                         ctx.__class__ = lfilesctx
                         return node
                     # Before commit, largefile standins have not had their
                     # contents updated to reflect the hash of their largefile.
                     # Do that here.
                     def commit(
                         self,
                         text=b"",
                         user=None,
                         date=None,
                         match=None,
                         force=False,
                         editor=False,
                         extra=None,
                     ):
                         if extra is None:
                             extra = {}
                         orig = super(lfilesrepo, self).commit
                         with self.wlock():
                             lfcommithook = self._lfcommithooks[-1]
                             match = lfcommithook(self, match)
                             result = orig(
                                 text=text,
                                 user=user,
                                 date=date,
                                 match=match,
                                 force=force,
                                 editor=editor,
                                 extra=extra,
                             )
                             return result
                     # TODO: _subdirlfs should be moved into "lfutil.py", because
                     # it is referred only from "lfutil.updatestandinsbymatch"
                     def _subdirlfs(self, files, lfiles):
                         """
                         Adjust matched file list
                         If we pass a directory to commit whose only committable files
                         are largefiles, the core commit code aborts before finding
                         the largefiles.
                         So we do the following:
                         For directories that only have largefiles as matches,
                         we explicitly add the largefiles to the match list and remove
                         the directory.
                         In other cases, we leave the match list unmodified.
                         """
                         actualfiles = []
                         dirs = []
                         regulars = []
                         for f in files:
                             if lfutil.isstandin(f + b'/'):
                                 raise error.Abort(
                                     _(b'file "%s" is a largefile standin') % f,
                                     hint=b'commit the largefile itself instead',
                                 )
                             # Scan directories
                             if self.wvfs.isdir(f):
                                 dirs.append(f)
                             else:
                                 regulars.append(f)
                         for f in dirs:
                             matcheddir = False
                             d = self.dirstate.normalize(f) + b'/'
                             # Check for matched normal files
                             for mf in regulars:
                                 if self.dirstate.normalize(mf).startswith(d):
                                     actualfiles.append(f)
                                     matcheddir = True
                                     break
                             if not matcheddir:
                                 # If no normal match, manually append
                                 # any matching largefiles
                                 for lf in lfiles:
                                     if self.dirstate.normalize(lf).startswith(d):
                                         actualfiles.append(lf)
                                         if not matcheddir:
                                             # There may still be normal files in the dir, so
                                             # add a directory to the list, which
                                             # forces status/dirstate to walk all files and
                                             # call the match function on the matcher, even
                                             # on case sensitive filesystems.
                                             actualfiles.append(b'.')
                                             matcheddir = True
                             # Nothing in dir, so readd it
                             # and let commit reject it
                             if not matcheddir:
                                 actualfiles.append(f)
                         # Always add normal files
                         actualfiles += regulars
                         return actualfiles
                 repo.__class__ = lfilesrepo
                 # stack of hooks being executed before committing.
                 # only last element ("_lfcommithooks[-1]") is used for each committing.
                 repo._lfcommithooks = [lfutil.updatestandinsbymatch]
                 # Stack of status writer functions taking "*msg, **opts" arguments
                 # like "ui.status()". Only last element ("_lfstatuswriters[-1]")
                 # is used to write status out.
                 repo._lfstatuswriters = [ui.status]
                 def prepushoutgoinghook(pushop):
                     """Push largefiles for pushop before pushing revisions."""
                     lfrevs = pushop.lfrevs
                     if lfrevs is None:
                         lfrevs = pushop.outgoing.missing
                     if lfrevs:
                         toupload = set()
                         addfunc = lambda fn, lfhash: toupload.add(lfhash)
                         lfutil.getlfilestoupload(pushop.repo, lfrevs, addfunc)
                         lfcommands.uploadlfiles(ui, pushop.repo, pushop.remote, toupload)
                 repo.prepushoutgoinghooks.add(b"largefiles", prepushoutgoinghook)
                 def checkrequireslfiles(ui, repo, **kwargs):
                     if b'largefiles' not in repo.requirements and any(
-                        lfutil.shortname + b'/' in f[0] for f in repo.store.datafiles()
+                        lfutil.shortname + b'/' in f[1] for f in repo.store.datafiles()
                     ):
                         repo.requirements.add(b'largefiles')
                         scmutil.writereporequirements(repo)
                 ui.setconfig(
                     b'hooks', b'changegroup.lfiles', checkrequireslfiles, b'largefiles'
                 )
                 ui.setconfig(b'hooks', b'commit.lfiles', checkrequireslfiles, b'largefiles')

hgext/narrow/narrowcommands.py

0 +1 -1

             # narrowcommands.py - command modifications for narrowhg extension
             #
             # Copyright 2017 Google, Inc.
             #
             # This software may be used and distributed according to the terms of the
             # GNU General Public License version 2 or any later version.
             from __future__ import absolute_import
             import itertools
             import os
             from mercurial.i18n import _
             from mercurial.node import (
                 hex,
                 nullid,
                 short,
             )
             from mercurial import (
                 bundle2,
                 cmdutil,
                 commands,
                 discovery,
                 encoding,
                 error,
                 exchange,
                 extensions,
                 hg,
                 narrowspec,
                 pathutil,
                 pycompat,
                 registrar,
                 repair,
                 repoview,
                 requirements,
                 sparse,
                 util,
                 wireprototypes,
             )
             table = {}
             command = registrar.command(table)
             def setup():
                 """Wraps user-facing mercurial commands with narrow-aware versions."""
                 entry = extensions.wrapcommand(commands.table, b'clone', clonenarrowcmd)
                 entry[1].append(
                     (b'', b'narrow', None, _(b"create a narrow clone of select files"))
                 )
                 entry[1].append(
                     (
                         b'',
                         b'depth',
                         b'',
                         _(b"limit the history fetched by distance from heads"),
                     )
                 )
                 entry[1].append((b'', b'narrowspec', b'', _(b"read narrowspecs from file")))
                 # TODO(durin42): unify sparse/narrow --include/--exclude logic a bit
                 if b'sparse' not in extensions.enabled():
                     entry[1].append(
                         (b'', b'include', [], _(b"specifically fetch this file/directory"))
                     )
                     entry[1].append(
                         (
                             b'',
                             b'exclude',
                             [],
                             _(b"do not fetch this file/directory, even if included"),
                         )
                     )
                 entry = extensions.wrapcommand(commands.table, b'pull', pullnarrowcmd)
                 entry[1].append(
                     (
                         b'',
                         b'depth',
                         b'',
                         _(b"limit the history fetched by distance from heads"),
                     )
                 )
                 extensions.wrapcommand(commands.table, b'archive', archivenarrowcmd)
             def clonenarrowcmd(orig, ui, repo, *args, **opts):
                 """Wraps clone command, so 'hg clone' first wraps localrepo.clone()."""
                 opts = pycompat.byteskwargs(opts)
                 wrappedextraprepare = util.nullcontextmanager()
                 narrowspecfile = opts[b'narrowspec']
                 if narrowspecfile:
                     filepath = os.path.join(encoding.getcwd(), narrowspecfile)
                     ui.status(_(b"reading narrowspec from '%s'\n") % filepath)
                     try:
                         fdata = util.readfile(filepath)
                     except IOError as inst:
                         raise error.Abort(
                             _(b"cannot read narrowspecs from '%s': %s")
                             % (filepath, encoding.strtolocal(inst.strerror))
                         )
                     includes, excludes, profiles = sparse.parseconfig(ui, fdata, b'narrow')
                     if profiles:
                         raise error.ConfigError(
                             _(
                                 b"cannot specify other files using '%include' in"
                                 b" narrowspec"
                             )
                         )
                     narrowspec.validatepatterns(includes)
                     narrowspec.validatepatterns(excludes)
                     # narrowspec is passed so we should assume that user wants narrow clone
                     opts[b'narrow'] = True
                     opts[b'include'].extend(includes)
                     opts[b'exclude'].extend(excludes)
                 if opts[b'narrow']:
                     def pullbundle2extraprepare_widen(orig, pullop, kwargs):
                         orig(pullop, kwargs)
                         if opts.get(b'depth'):
                             kwargs[b'depth'] = opts[b'depth']
                     wrappedextraprepare = extensions.wrappedfunction(
                         exchange, b'_pullbundle2extraprepare', pullbundle2extraprepare_widen
                     )
                 with wrappedextraprepare:
                     return orig(ui, repo, *args, **pycompat.strkwargs(opts))
             def pullnarrowcmd(orig, ui, repo, *args, **opts):
                 """Wraps pull command to allow modifying narrow spec."""
                 wrappedextraprepare = util.nullcontextmanager()
                 if requirements.NARROW_REQUIREMENT in repo.requirements:
                     def pullbundle2extraprepare_widen(orig, pullop, kwargs):
                         orig(pullop, kwargs)
                         if opts.get('depth'):
                             kwargs[b'depth'] = opts['depth']
                     wrappedextraprepare = extensions.wrappedfunction(
                         exchange, b'_pullbundle2extraprepare', pullbundle2extraprepare_widen
                     )
                 with wrappedextraprepare:
                     return orig(ui, repo, *args, **opts)
             def archivenarrowcmd(orig, ui, repo, *args, **opts):
                 """Wraps archive command to narrow the default includes."""
                 if requirements.NARROW_REQUIREMENT in repo.requirements:
                     repo_includes, repo_excludes = repo.narrowpats
                     includes = set(opts.get('include', []))
                     excludes = set(opts.get('exclude', []))
                     includes, excludes, unused_invalid = narrowspec.restrictpatterns(
                         includes, excludes, repo_includes, repo_excludes
                     )
                     if includes:
                         opts['include'] = includes
                     if excludes:
                         opts['exclude'] = excludes
                 return orig(ui, repo, *args, **opts)
             def pullbundle2extraprepare(orig, pullop, kwargs):
                 repo = pullop.repo
                 if requirements.NARROW_REQUIREMENT not in repo.requirements:
                     return orig(pullop, kwargs)
                 if wireprototypes.NARROWCAP not in pullop.remote.capabilities():
                     raise error.Abort(_(b"server does not support narrow clones"))
                 orig(pullop, kwargs)
                 kwargs[b'narrow'] = True
                 include, exclude = repo.narrowpats
                 kwargs[b'oldincludepats'] = include
                 kwargs[b'oldexcludepats'] = exclude
                 if include:
                     kwargs[b'includepats'] = include
                 if exclude:
                     kwargs[b'excludepats'] = exclude
                 # calculate known nodes only in ellipses cases because in non-ellipses cases
                 # we have all the nodes
                 if wireprototypes.ELLIPSESCAP1 in pullop.remote.capabilities():
                     kwargs[b'known'] = [
                         hex(ctx.node())
                         for ctx in repo.set(b'::%ln', pullop.common)
                         if ctx.node() != nullid
                     ]
                     if not kwargs[b'known']:
                         # Mercurial serializes an empty list as '' and deserializes it as
                         # [''], so delete it instead to avoid handling the empty string on
                         # the server.
                         del kwargs[b'known']
             extensions.wrapfunction(
                 exchange, b'_pullbundle2extraprepare', pullbundle2extraprepare
             )
             def _narrow(
                 ui,
                 repo,
                 remote,
                 commoninc,
                 oldincludes,
                 oldexcludes,
                 newincludes,
                 newexcludes,
                 force,
                 backup,
             ):
                 oldmatch = narrowspec.match(repo.root, oldincludes, oldexcludes)
                 newmatch = narrowspec.match(repo.root, newincludes, newexcludes)
                 # This is essentially doing "hg outgoing" to find all local-only
                 # commits. We will then check that the local-only commits don't
                 # have any changes to files that will be untracked.
                 unfi = repo.unfiltered()
                 outgoing = discovery.findcommonoutgoing(unfi, remote, commoninc=commoninc)
                 ui.status(_(b'looking for local changes to affected paths\n'))
                 localnodes = []
                 for n in itertools.chain(outgoing.missing, outgoing.excluded):
                     if any(oldmatch(f) and not newmatch(f) for f in unfi[n].files()):
                         localnodes.append(n)
                 revstostrip = unfi.revs(b'descendants(%ln)', localnodes)
                 hiddenrevs = repoview.filterrevs(repo, b'visible')
                 visibletostrip = list(
                     repo.changelog.node(r) for r in (revstostrip - hiddenrevs)
                 )
                 if visibletostrip:
                     ui.status(
                         _(
                             b'The following changeset(s) or their ancestors have '
                             b'local changes not on the remote:\n'
                         )
                     )
                     maxnodes = 10
                     if ui.verbose or len(visibletostrip) <= maxnodes:
                         for n in visibletostrip:
                             ui.status(b'%s\n' % short(n))
                     else:
                         for n in visibletostrip[:maxnodes]:
                             ui.status(b'%s\n' % short(n))
                         ui.status(
                             _(b'...and %d more, use --verbose to list all\n')
                             % (len(visibletostrip) - maxnodes)
                         )
                     if not force:
                         raise error.StateError(
                             _(b'local changes found'),
                             hint=_(b'use --force-delete-local-changes to ignore'),
                         )
                 with ui.uninterruptible():
                     if revstostrip:
                         tostrip = [unfi.changelog.node(r) for r in revstostrip]
                         if repo[b'.'].node() in tostrip:
                             # stripping working copy, so move to a different commit first
                             urev = max(
                                 repo.revs(
                                     b'(::%n) - %ln + null',
                                     repo[b'.'].node(),
                                     visibletostrip,
                                 )
                             )
                             hg.clean(repo, urev)
                         overrides = {(b'devel', b'strip-obsmarkers'): False}
                         with ui.configoverride(overrides, b'narrow'):
                             repair.strip(ui, unfi, tostrip, topic=b'narrow', backup=backup)
                     todelete = []
-                    for f, f2, size in repo.store.datafiles():
+                    for t, f, f2, size in repo.store.datafiles():
                         if f.startswith(b'data/'):
                             file = f[5:-2]
                             if not newmatch(file):
                                 todelete.append(f)
                         elif f.startswith(b'meta/'):
                             dir = f[5:-13]
                             dirs = sorted(pathutil.dirs({dir})) + [dir]
                             include = True
                             for d in dirs:
                                 visit = newmatch.visitdir(d)
                                 if not visit:
                                     include = False
                                     break
                                 if visit == b'all':
                                     break
                             if not include:
                                 todelete.append(f)
                     repo.destroying()
                     with repo.transaction(b'narrowing'):
                         # Update narrowspec before removing revlogs, so repo won't be
                         # corrupt in case of crash
                         repo.setnarrowpats(newincludes, newexcludes)
                         for f in todelete:
                             ui.status(_(b'deleting %s\n') % f)
                             util.unlinkpath(repo.svfs.join(f))
                             repo.store.markremoved(f)
                         narrowspec.updateworkingcopy(repo, assumeclean=True)
                         narrowspec.copytoworkingcopy(repo)
                     repo.destroyed()
             def _widen(
                 ui,
                 repo,
                 remote,
                 commoninc,
                 oldincludes,
                 oldexcludes,
                 newincludes,
                 newexcludes,
             ):
                 # for now we assume that if a server has ellipses enabled, we will be
                 # exchanging ellipses nodes. In future we should add ellipses as a client
                 # side requirement (maybe) to distinguish a client is shallow or not and
                 # then send that information to server whether we want ellipses or not.
                 # Theoretically a non-ellipses repo should be able to use narrow
                 # functionality from an ellipses enabled server
                 remotecap = remote.capabilities()
                 ellipsesremote = any(
                     cap in remotecap for cap in wireprototypes.SUPPORTED_ELLIPSESCAP
                 )
                 # check whether we are talking to a server which supports old version of
                 # ellipses capabilities
                 isoldellipses = (
                     ellipsesremote
                     and wireprototypes.ELLIPSESCAP1 in remotecap
                     and wireprototypes.ELLIPSESCAP not in remotecap
                 )
                 def pullbundle2extraprepare_widen(orig, pullop, kwargs):
                     orig(pullop, kwargs)
                     # The old{in,ex}cludepats have already been set by orig()
                     kwargs[b'includepats'] = newincludes
                     kwargs[b'excludepats'] = newexcludes
                 wrappedextraprepare = extensions.wrappedfunction(
                     exchange, b'_pullbundle2extraprepare', pullbundle2extraprepare_widen
                 )
                 # define a function that narrowbundle2 can call after creating the
                 # backup bundle, but before applying the bundle from the server
                 def setnewnarrowpats():
                     repo.setnarrowpats(newincludes, newexcludes)
                 repo.setnewnarrowpats = setnewnarrowpats
                 # silence the devel-warning of applying an empty changegroup
                 overrides = {(b'devel', b'all-warnings'): False}
                 common = commoninc[0]
                 with ui.uninterruptible():
                     if ellipsesremote:
                         ds = repo.dirstate
                         p1, p2 = ds.p1(), ds.p2()
                         with ds.parentchange():
                             ds.setparents(nullid, nullid)
                     if isoldellipses:
                         with wrappedextraprepare:
                             exchange.pull(repo, remote, heads=common)
                     else:
                         known = []
                         if ellipsesremote:
                             known = [
                                 ctx.node()
                                 for ctx in repo.set(b'::%ln', common)
                                 if ctx.node() != nullid
                             ]
                         with remote.commandexecutor() as e:
                             bundle = e.callcommand(
                                 b'narrow_widen',
                                 {
                                     b'oldincludes': oldincludes,
                                     b'oldexcludes': oldexcludes,
                                     b'newincludes': newincludes,
                                     b'newexcludes': newexcludes,
                                     b'cgversion': b'03',
                                     b'commonheads': common,
                                     b'known': known,
                                     b'ellipses': ellipsesremote,
                                 },
                             ).result()
                         trmanager = exchange.transactionmanager(
                             repo, b'widen', remote.url()
                         )
                         with trmanager, repo.ui.configoverride(overrides, b'widen'):
                             op = bundle2.bundleoperation(
                                 repo, trmanager.transaction, source=b'widen'
                             )
                             # TODO: we should catch error.Abort here
                             bundle2.processbundle(repo, bundle, op=op)
                     if ellipsesremote:
                         with ds.parentchange():
                             ds.setparents(p1, p2)
                     with repo.transaction(b'widening'):
                         repo.setnewnarrowpats()
                         narrowspec.updateworkingcopy(repo)
                         narrowspec.copytoworkingcopy(repo)
             # TODO(rdamazio): Make new matcher format and update description
             @command(
                 b'tracked',
                 [
                     (b'', b'addinclude', [], _(b'new paths to include')),
                     (b'', b'removeinclude', [], _(b'old paths to no longer include')),
                     (
                         b'',
                         b'auto-remove-includes',
                         False,
                         _(b'automatically choose unused includes to remove'),
                     ),
                     (b'', b'addexclude', [], _(b'new paths to exclude')),
                     (b'', b'import-rules', b'', _(b'import narrowspecs from a file')),
                     (b'', b'removeexclude', [], _(b'old paths to no longer exclude')),
                     (
                         b'',
                         b'clear',
                         False,
                         _(b'whether to replace the existing narrowspec'),
                     ),
                     (
                         b'',
                         b'force-delete-local-changes',
                         False,
                         _(b'forces deletion of local changes when narrowing'),
                     ),
                     (
                         b'',
                         b'backup',
                         True,
                         _(b'back up local changes when narrowing'),
                     ),
                     (
                         b'',
                         b'update-working-copy',
                         False,
                         _(b'update working copy when the store has changed'),
                     ),
                 ]
                 + commands.remoteopts,
                 _(b'[OPTIONS]... [REMOTE]'),
                 inferrepo=True,
                 helpcategory=command.CATEGORY_MAINTENANCE,
             )
             def trackedcmd(ui, repo, remotepath=None, *pats, **opts):
                 """show or change the current narrowspec
                 With no argument, shows the current narrowspec entries, one per line. Each
                 line will be prefixed with 'I' or 'X' for included or excluded patterns,
                 respectively.
                 The narrowspec is comprised of expressions to match remote files and/or
                 directories that should be pulled into your client.
                 The narrowspec has *include* and *exclude* expressions, with excludes always
                 trumping includes: that is, if a file matches an exclude expression, it will
                 be excluded even if it also matches an include expression.
                 Excluding files that were never included has no effect.
                 Each included or excluded entry is in the format described by
                 'hg help patterns'.
                 The options allow you to add or remove included and excluded expressions.
                 If --clear is specified, then all previous includes and excludes are DROPPED
                 and replaced by the new ones specified to --addinclude and --addexclude.
                 If --clear is specified without any further options, the narrowspec will be
                 empty and will not match any files.
                 If --auto-remove-includes is specified, then those includes that don't match
                 any files modified by currently visible local commits (those not shared by
                 the remote) will be added to the set of explicitly specified includes to
                 remove.
                 --import-rules accepts a path to a file containing rules, allowing you to
                 add --addinclude, --addexclude rules in bulk. Like the other include and
                 exclude switches, the changes are applied immediately.
                 """
                 opts = pycompat.byteskwargs(opts)
                 if requirements.NARROW_REQUIREMENT not in repo.requirements:
                     raise error.InputError(
                         _(
                             b'the tracked command is only supported on '
                             b'repositories cloned with --narrow'
                         )
                     )
                 # Before supporting, decide whether it "hg tracked --clear" should mean
                 # tracking no paths or all paths.
                 if opts[b'clear']:
                     raise error.InputError(_(b'the --clear option is not yet supported'))
                 # import rules from a file
                 newrules = opts.get(b'import_rules')
                 if newrules:
                     try:
                         filepath = os.path.join(encoding.getcwd(), newrules)
                         fdata = util.readfile(filepath)
                     except IOError as inst:
                         raise error.StorageError(
                             _(b"cannot read narrowspecs from '%s': %s")
                             % (filepath, encoding.strtolocal(inst.strerror))
                         )
                     includepats, excludepats, profiles = sparse.parseconfig(
                         ui, fdata, b'narrow'
                     )
                     if profiles:
                         raise error.InputError(
                             _(
                                 b"including other spec files using '%include' "
                                 b"is not supported in narrowspec"
                             )
                         )
                     opts[b'addinclude'].extend(includepats)
                     opts[b'addexclude'].extend(excludepats)
                 addedincludes = narrowspec.parsepatterns(opts[b'addinclude'])
                 removedincludes = narrowspec.parsepatterns(opts[b'removeinclude'])
                 addedexcludes = narrowspec.parsepatterns(opts[b'addexclude'])
                 removedexcludes = narrowspec.parsepatterns(opts[b'removeexclude'])
                 autoremoveincludes = opts[b'auto_remove_includes']
                 update_working_copy = opts[b'update_working_copy']
                 only_show = not (
                     addedincludes
                     or removedincludes
                     or addedexcludes
                     or removedexcludes
                     or newrules
                     or autoremoveincludes
                     or update_working_copy
                 )
                 oldincludes, oldexcludes = repo.narrowpats
                 # filter the user passed additions and deletions into actual additions and
                 # deletions of excludes and includes
                 addedincludes -= oldincludes
                 removedincludes &= oldincludes
                 addedexcludes -= oldexcludes
                 removedexcludes &= oldexcludes
                 widening = addedincludes or removedexcludes
                 narrowing = removedincludes or addedexcludes
                 # Only print the current narrowspec.
                 if only_show:
                     ui.pager(b'tracked')
                     fm = ui.formatter(b'narrow', opts)
                     for i in sorted(oldincludes):
                         fm.startitem()
                         fm.write(b'status', b'%s ', b'I', label=b'narrow.included')
                         fm.write(b'pat', b'%s\n', i, label=b'narrow.included')
                     for i in sorted(oldexcludes):
                         fm.startitem()
                         fm.write(b'status', b'%s ', b'X', label=b'narrow.excluded')
                         fm.write(b'pat', b'%s\n', i, label=b'narrow.excluded')
                     fm.end()
                     return 0
                 if update_working_copy:
                     with repo.wlock(), repo.lock(), repo.transaction(b'narrow-wc'):
                         narrowspec.updateworkingcopy(repo)
                         narrowspec.copytoworkingcopy(repo)
                     return 0
                 if not (widening or narrowing or autoremoveincludes):
                     ui.status(_(b"nothing to widen or narrow\n"))
                     return 0
                 with repo.wlock(), repo.lock():
                     cmdutil.bailifchanged(repo)
                     # Find the revisions we have in common with the remote. These will
                     # be used for finding local-only changes for narrowing. They will
                     # also define the set of revisions to update for widening.
                     remotepath = ui.expandpath(remotepath or b'default')
                     url, branches = hg.parseurl(remotepath)
                     ui.status(_(b'comparing with %s\n') % util.hidepassword(url))
                     remote = hg.peer(repo, opts, url)
                     try:
                         # check narrow support before doing anything if widening needs to be
                         # performed. In future we should also abort if client is ellipses and
                         # server does not support ellipses
                         if (
                             widening
                             and wireprototypes.NARROWCAP not in remote.capabilities()
                         ):
                             raise error.Abort(_(b"server does not support narrow clones"))
                         commoninc = discovery.findcommonincoming(repo, remote)
                         if autoremoveincludes:
                             outgoing = discovery.findcommonoutgoing(
                                 repo, remote, commoninc=commoninc
                             )
                             ui.status(_(b'looking for unused includes to remove\n'))
                             localfiles = set()
                             for n in itertools.chain(outgoing.missing, outgoing.excluded):
                                 localfiles.update(repo[n].files())
                             suggestedremovals = []
                             for include in sorted(oldincludes):
                                 match = narrowspec.match(repo.root, [include], oldexcludes)
                                 if not any(match(f) for f in localfiles):
                                     suggestedremovals.append(include)
                             if suggestedremovals:
                                 for s in suggestedremovals:
                                     ui.status(b'%s\n' % s)
                                 if (
                                     ui.promptchoice(
                                         _(
                                             b'remove these unused includes (yn)?'
                                             b'$$ &Yes $$ &No'
                                         )
                                     )
                                     == 0
                                 ):
                                     removedincludes.update(suggestedremovals)
                                     narrowing = True
                             else:
                                 ui.status(_(b'found no unused includes\n'))
                         if narrowing:
                             newincludes = oldincludes - removedincludes
                             newexcludes = oldexcludes | addedexcludes
                             _narrow(
                                 ui,
                                 repo,
                                 remote,
                                 commoninc,
                                 oldincludes,
                                 oldexcludes,
                                 newincludes,
                                 newexcludes,
                                 opts[b'force_delete_local_changes'],
                                 opts[b'backup'],
                             )
                             # _narrow() updated the narrowspec and _widen() below needs to
                             # use the updated values as its base (otherwise removed includes
                             # and addedexcludes will be lost in the resulting narrowspec)
                             oldincludes = newincludes
                             oldexcludes = newexcludes
                         if widening:
                             newincludes = oldincludes | addedincludes
                             newexcludes = oldexcludes - removedexcludes
                             _widen(
                                 ui,
                                 repo,
                                 remote,
                                 commoninc,
                                 oldincludes,
                                 oldexcludes,
                                 newincludes,
                                 newexcludes,
                             )
                     finally:
                         remote.close()
                 return 0

hgext/remotefilelog/contentstore.py

0 +1 -1

             from __future__ import absolute_import
             import threading
             from mercurial.node import hex, nullid
             from mercurial.pycompat import getattr
             from mercurial import (
                 mdiff,
                 pycompat,
                 revlog,
             )
             from . import (
                 basestore,
                 constants,
                 shallowutil,
             )
             class ChainIndicies(object):
                 """A static class for easy reference to the delta chain indicies."""
                 # The filename of this revision delta
                 NAME = 0
                 # The mercurial file node for this revision delta
                 NODE = 1
                 # The filename of the delta base's revision. This is useful when delta
                 # between different files (like in the case of a move or copy, we can delta
                 # against the original file content).
                 BASENAME = 2
                 # The mercurial file node for the delta base revision. This is the nullid if
                 # this delta is a full text.
                 BASENODE = 3
                 # The actual delta or full text data.
                 DATA = 4
             class unioncontentstore(basestore.baseunionstore):
                 def __init__(self, *args, **kwargs):
                     super(unioncontentstore, self).__init__(*args, **kwargs)
                     self.stores = args
                     self.writestore = kwargs.get('writestore')
                     # If allowincomplete==True then the union store can return partial
                     # delta chains, otherwise it will throw a KeyError if a full
                     # deltachain can't be found.
                     self.allowincomplete = kwargs.get('allowincomplete', False)
                 def get(self, name, node):
                     """Fetches the full text revision contents of the given name+node pair.
                     If the full text doesn't exist, throws a KeyError.
                     Under the hood, this uses getdeltachain() across all the stores to build
                     up a full chain to produce the full text.
                     """
                     chain = self.getdeltachain(name, node)
                     if chain[-1][ChainIndicies.BASENODE] != nullid:
                         # If we didn't receive a full chain, throw
                         raise KeyError((name, hex(node)))
                     # The last entry in the chain is a full text, so we start our delta
                     # applies with that.
                     fulltext = chain.pop()[ChainIndicies.DATA]
                     text = fulltext
                     while chain:
                         delta = chain.pop()[ChainIndicies.DATA]
                         text = mdiff.patches(text, [delta])
                     return text
                 @basestore.baseunionstore.retriable
                 def getdelta(self, name, node):
                     """Return the single delta entry for the given name/node pair."""
                     for store in self.stores:
                         try:
                             return store.getdelta(name, node)
                         except KeyError:
                             pass
                     raise KeyError((name, hex(node)))
                 def getdeltachain(self, name, node):
                     """Returns the deltachain for the given name/node pair.
                     Returns an ordered list of:
                       [(name, node, deltabasename, deltabasenode, deltacontent),...]
                     where the chain is terminated by a full text entry with a nullid
                     deltabasenode.
                     """
                     chain = self._getpartialchain(name, node)
                     while chain[-1][ChainIndicies.BASENODE] != nullid:
                         x, x, deltabasename, deltabasenode, x = chain[-1]
                         try:
                             morechain = self._getpartialchain(deltabasename, deltabasenode)
                             chain.extend(morechain)
                         except KeyError:
                             # If we allow incomplete chains, don't throw.
                             if not self.allowincomplete:
                                 raise
                             break
                     return chain
                 @basestore.baseunionstore.retriable
                 def getmeta(self, name, node):
                     """Returns the metadata dict for given node."""
                     for store in self.stores:
                         try:
                             return store.getmeta(name, node)
                         except KeyError:
                             pass
                     raise KeyError((name, hex(node)))
                 def getmetrics(self):
                     metrics = [s.getmetrics() for s in self.stores]
                     return shallowutil.sumdicts(*metrics)
                 @basestore.baseunionstore.retriable
                 def _getpartialchain(self, name, node):
                     """Returns a partial delta chain for the given name/node pair.
                     A partial chain is a chain that may not be terminated in a full-text.
                     """
                     for store in self.stores:
                         try:
                             return store.getdeltachain(name, node)
                         except KeyError:
                             pass
                     raise KeyError((name, hex(node)))
                 def add(self, name, node, data):
                     raise RuntimeError(
                         b"cannot add content only to remotefilelog contentstore"
                     )
                 def getmissing(self, keys):
                     missing = keys
                     for store in self.stores:
                         if missing:
                             missing = store.getmissing(missing)
                     return missing
                 def addremotefilelognode(self, name, node, data):
                     if self.writestore:
                         self.writestore.addremotefilelognode(name, node, data)
                     else:
                         raise RuntimeError(b"no writable store configured")
                 def markledger(self, ledger, options=None):
                     for store in self.stores:
                         store.markledger(ledger, options)
             class remotefilelogcontentstore(basestore.basestore):
                 def __init__(self, *args, **kwargs):
                     super(remotefilelogcontentstore, self).__init__(*args, **kwargs)
                     self._threaddata = threading.local()
                 def get(self, name, node):
                     # return raw revision text
                     data = self._getdata(name, node)
                     offset, size, flags = shallowutil.parsesizeflags(data)
                     content = data[offset : offset + size]
                     ancestormap = shallowutil.ancestormap(data)
                     p1, p2, linknode, copyfrom = ancestormap[node]
                     copyrev = None
                     if copyfrom:
                         copyrev = hex(p1)
                     self._updatemetacache(node, size, flags)
                     # lfs tracks renames in its own metadata, remove hg copy metadata,
                     # because copy metadata will be re-added by lfs flag processor.
                     if flags & revlog.REVIDX_EXTSTORED:
                         copyrev = copyfrom = None
                     revision = shallowutil.createrevlogtext(content, copyfrom, copyrev)
                     return revision
                 def getdelta(self, name, node):
                     # Since remotefilelog content stores only contain full texts, just
                     # return that.
                     revision = self.get(name, node)
                     return revision, name, nullid, self.getmeta(name, node)
                 def getdeltachain(self, name, node):
                     # Since remotefilelog content stores just contain full texts, we return
                     # a fake delta chain that just consists of a single full text revision.
                     # The nullid in the deltabasenode slot indicates that the revision is a
                     # fulltext.
                     revision = self.get(name, node)
                     return [(name, node, None, nullid, revision)]
                 def getmeta(self, name, node):
                     self._sanitizemetacache()
                     if node != self._threaddata.metacache[0]:
                         data = self._getdata(name, node)
                         offset, size, flags = shallowutil.parsesizeflags(data)
                         self._updatemetacache(node, size, flags)
                     return self._threaddata.metacache[1]
                 def add(self, name, node, data):
                     raise RuntimeError(
                         b"cannot add content only to remotefilelog contentstore"
                     )
                 def _sanitizemetacache(self):
                     metacache = getattr(self._threaddata, 'metacache', None)
                     if metacache is None:
                         self._threaddata.metacache = (None, None)  # (node, meta)
                 def _updatemetacache(self, node, size, flags):
                     self._sanitizemetacache()
                     if node == self._threaddata.metacache[0]:
                         return
                     meta = {constants.METAKEYFLAG: flags, constants.METAKEYSIZE: size}
                     self._threaddata.metacache = (node, meta)
             class remotecontentstore(object):
                 def __init__(self, ui, fileservice, shared):
                     self._fileservice = fileservice
                     # type(shared) is usually remotefilelogcontentstore
                     self._shared = shared
                 def get(self, name, node):
                     self._fileservice.prefetch(
                         [(name, hex(node))], force=True, fetchdata=True
                     )
                     return self._shared.get(name, node)
                 def getdelta(self, name, node):
                     revision = self.get(name, node)
                     return revision, name, nullid, self._shared.getmeta(name, node)
                 def getdeltachain(self, name, node):
                     # Since our remote content stores just contain full texts, we return a
                     # fake delta chain that just consists of a single full text revision.
                     # The nullid in the deltabasenode slot indicates that the revision is a
                     # fulltext.
                     revision = self.get(name, node)
                     return [(name, node, None, nullid, revision)]
                 def getmeta(self, name, node):
                     self._fileservice.prefetch(
                         [(name, hex(node))], force=True, fetchdata=True
                     )
                     return self._shared.getmeta(name, node)
                 def add(self, name, node, data):
                     raise RuntimeError(b"cannot add to a remote store")
                 def getmissing(self, keys):
                     return keys
                 def markledger(self, ledger, options=None):
                     pass
             class manifestrevlogstore(object):
                 def __init__(self, repo):
                     self._store = repo.store
                     self._svfs = repo.svfs
                     self._revlogs = dict()
                     self._cl = revlog.revlog(self._svfs, b'00changelog.i')
                     self._repackstartlinkrev = 0
                 def get(self, name, node):
                     return self._revlog(name).rawdata(node)
                 def getdelta(self, name, node):
                     revision = self.get(name, node)
                     return revision, name, nullid, self.getmeta(name, node)
                 def getdeltachain(self, name, node):
                     revision = self.get(name, node)
                     return [(name, node, None, nullid, revision)]
                 def getmeta(self, name, node):
                     rl = self._revlog(name)
                     rev = rl.rev(node)
                     return {
                         constants.METAKEYFLAG: rl.flags(rev),
                         constants.METAKEYSIZE: rl.rawsize(rev),
                     }
                 def getancestors(self, name, node, known=None):
                     if known is None:
                         known = set()
                     if node in known:
                         return []
                     rl = self._revlog(name)
                     ancestors = {}
                     missing = {node}
                     for ancrev in rl.ancestors([rl.rev(node)], inclusive=True):
                         ancnode = rl.node(ancrev)
                         missing.discard(ancnode)
                         p1, p2 = rl.parents(ancnode)
                         if p1 != nullid and p1 not in known:
                             missing.add(p1)
                         if p2 != nullid and p2 not in known:
                             missing.add(p2)
                         linknode = self._cl.node(rl.linkrev(ancrev))
                         ancestors[rl.node(ancrev)] = (p1, p2, linknode, b'')
                         if not missing:
                             break
                     return ancestors
                 def getnodeinfo(self, name, node):
                     cl = self._cl
                     rl = self._revlog(name)
                     parents = rl.parents(node)
                     linkrev = rl.linkrev(rl.rev(node))
                     return (parents[0], parents[1], cl.node(linkrev), None)
                 def add(self, *args):
                     raise RuntimeError(b"cannot add to a revlog store")
                 def _revlog(self, name):
                     rl = self._revlogs.get(name)
                     if rl is None:
                         revlogname = b'00manifesttree.i'
                         if name != b'':
                             revlogname = b'meta/%s/00manifest.i' % name
                         rl = revlog.revlog(self._svfs, revlogname)
                         self._revlogs[name] = rl
                     return rl
                 def getmissing(self, keys):
                     missing = []
                     for name, node in keys:
                         mfrevlog = self._revlog(name)
                         if node not in mfrevlog.nodemap:
                             missing.append((name, node))
                     return missing
                 def setrepacklinkrevrange(self, startrev, endrev):
                     self._repackstartlinkrev = startrev
                     self._repackendlinkrev = endrev
                 def markledger(self, ledger, options=None):
                     if options and options.get(constants.OPTION_PACKSONLY):
                         return
                     treename = b''
                     rl = revlog.revlog(self._svfs, b'00manifesttree.i')
                     startlinkrev = self._repackstartlinkrev
                     endlinkrev = self._repackendlinkrev
                     for rev in pycompat.xrange(len(rl) - 1, -1, -1):
                         linkrev = rl.linkrev(rev)
                         if linkrev < startlinkrev:
                             break
                         if linkrev > endlinkrev:
                             continue
                         node = rl.node(rev)
                         ledger.markdataentry(self, treename, node)
                         ledger.markhistoryentry(self, treename, node)
-                    for path, encoded, size in self._store.datafiles():
+                    for t, path, encoded, size in self._store.datafiles():
                         if path[:5] != b'meta/' or path[-2:] != b'.i':
                             continue
                         treename = path[5 : -len(b'/00manifest.i')]
                         rl = revlog.revlog(self._svfs, path)
                         for rev in pycompat.xrange(len(rl) - 1, -1, -1):
                             linkrev = rl.linkrev(rev)
                             if linkrev < startlinkrev:
                                 break
                             if linkrev > endlinkrev:
                                 continue
                             node = rl.node(rev)
                             ledger.markdataentry(self, treename, node)
                             ledger.markhistoryentry(self, treename, node)
                 def cleanup(self, ledger):
                     pass

hgext/remotefilelog/remotefilelogserver.py

0 +7 -5

             # remotefilelogserver.py - server logic for a remotefilelog server
             #
             # Copyright 2013 Facebook, Inc.
             #
             # This software may be used and distributed according to the terms of the
             # GNU General Public License version 2 or any later version.
             from __future__ import absolute_import
             import errno
             import os
             import stat
             import time
             import zlib
             from mercurial.i18n import _
             from mercurial.node import bin, hex, nullid
             from mercurial.pycompat import open
             from mercurial import (
                 changegroup,
                 changelog,
                 context,
                 error,
                 extensions,
                 match,
                 pycompat,
                 scmutil,
                 store,
                 streamclone,
                 util,
                 wireprotoserver,
                 wireprototypes,
                 wireprotov1server,
             )
             from . import (
                 constants,
                 shallowutil,
             )
             _sshv1server = wireprotoserver.sshv1protocolhandler
             def setupserver(ui, repo):
                 """Sets up a normal Mercurial repo so it can serve files to shallow repos."""
                 onetimesetup(ui)
                 # don't send files to shallow clients during pulls
                 def generatefiles(
                     orig, self, changedfiles, linknodes, commonrevs, source, *args, **kwargs
                 ):
                     caps = self._bundlecaps or []
                     if constants.BUNDLE2_CAPABLITY in caps:
                         # only send files that don't match the specified patterns
                         includepattern = None
                         excludepattern = None
                         for cap in self._bundlecaps or []:
                             if cap.startswith(b"includepattern="):
                                 includepattern = cap[len(b"includepattern=") :].split(b'\0')
                             elif cap.startswith(b"excludepattern="):
                                 excludepattern = cap[len(b"excludepattern=") :].split(b'\0')
                         m = match.always()
                         if includepattern or excludepattern:
                             m = match.match(
                                 repo.root, b'', None, includepattern, excludepattern
                             )
                         changedfiles = list([f for f in changedfiles if not m(f)])
                     return orig(
                         self, changedfiles, linknodes, commonrevs, source, *args, **kwargs
                     )
                 extensions.wrapfunction(
                     changegroup.cgpacker, b'generatefiles', generatefiles
                 )
             onetime = False
             def onetimesetup(ui):
                 """Configures the wireprotocol for both clients and servers."""
                 global onetime
                 if onetime:
                     return
                 onetime = True
                 # support file content requests
                 wireprotov1server.wireprotocommand(
                     b'x_rfl_getflogheads', b'path', permission=b'pull'
                 )(getflogheads)
                 wireprotov1server.wireprotocommand(
                     b'x_rfl_getfiles', b'', permission=b'pull'
                 )(getfiles)
                 wireprotov1server.wireprotocommand(
                     b'x_rfl_getfile', b'file node', permission=b'pull'
                 )(getfile)
                 class streamstate(object):
                     match = None
                     shallowremote = False
                     noflatmf = False
                 state = streamstate()
                 def stream_out_shallow(repo, proto, other):
                     includepattern = None
                     excludepattern = None
                     raw = other.get(b'includepattern')
                     if raw:
                         includepattern = raw.split(b'\0')
                     raw = other.get(b'excludepattern')
                     if raw:
                         excludepattern = raw.split(b'\0')
                     oldshallow = state.shallowremote
                     oldmatch = state.match
                     oldnoflatmf = state.noflatmf
                     try:
                         state.shallowremote = True
                         state.match = match.always()
                         state.noflatmf = other.get(b'noflatmanifest') == b'True'
                         if includepattern or excludepattern:
                             state.match = match.match(
                                 repo.root, b'', None, includepattern, excludepattern
                             )
                         streamres = wireprotov1server.stream(repo, proto)
                         # Force the first value to execute, so the file list is computed
                         # within the try/finally scope
                         first = next(streamres.gen)
                         second = next(streamres.gen)
                         def gen():
                             yield first
                             yield second
                             for value in streamres.gen:
                                 yield value
                         return wireprototypes.streamres(gen())
                     finally:
                         state.shallowremote = oldshallow
                         state.match = oldmatch
                         state.noflatmf = oldnoflatmf
                 wireprotov1server.commands[b'stream_out_shallow'] = (
                     stream_out_shallow,
                     b'*',
                 )
                 # don't clone filelogs to shallow clients
                 def _walkstreamfiles(orig, repo, matcher=None):
                     if state.shallowremote:
                         # if we are shallow ourselves, stream our local commits
                         if shallowutil.isenabled(repo):
                             striplen = len(repo.store.path) + 1
                             readdir = repo.store.rawvfs.readdir
                             visit = [os.path.join(repo.store.path, b'data')]
                             while visit:
                                 p = visit.pop()
                                 for f, kind, st in readdir(p, stat=True):
                                     fp = p + b'/' + f
                                     if kind == stat.S_IFREG:
                                         if not fp.endswith(b'.i') and not fp.endswith(
                                             b'.d'
                                         ):
                                             n = util.pconvert(fp[striplen:])
-                                            yield (store.decodedir(n), n, st.st_size)
+                                            d = store.decodedir(n)
+                                            t = store.FILETYPE_OTHER
+                                            yield (t, d, n, st.st_size)
                                     if kind == stat.S_IFDIR:
                                         visit.append(fp)
                         if scmutil.istreemanifest(repo):
-                            for (u, e, s) in repo.store.datafiles():
+                            for (t, u, e, s) in repo.store.datafiles():
                                 if u.startswith(b'meta/') and (
                                     u.endswith(b'.i') or u.endswith(b'.d')
                                 ):
-                                    yield (u, e, s)
+                                    yield (t, u, e, s)
                         # Return .d and .i files that do not match the shallow pattern
                         match = state.match
                         if match and not match.always():
-                            for (u, e, s) in repo.store.datafiles():
+                            for (t, u, e, s) in repo.store.datafiles():
                                 f = u[5:-2]  # trim data/...  and .i/.d
                                 if not state.match(f):
-                                    yield (u, e, s)
+                                    yield (t, u, e, s)
                         for x in repo.store.topfiles():
                             if state.noflatmf and x[0][:11] == b'00manifest.':
                                 continue
                             yield x
                     elif shallowutil.isenabled(repo):
                         # don't allow cloning from a shallow repo to a full repo
                         # since it would require fetching every version of every
                         # file in order to create the revlogs.
                         raise error.Abort(
                             _(b"Cannot clone from a shallow repo to a full repo.")
                         )
                     else:
                         for x in orig(repo, matcher):
                             yield x
                 extensions.wrapfunction(streamclone, b'_walkstreamfiles', _walkstreamfiles)
                 # expose remotefilelog capabilities
                 def _capabilities(orig, repo, proto):
                     caps = orig(repo, proto)
                     if shallowutil.isenabled(repo) or ui.configbool(
                         b'remotefilelog', b'server'
                     ):
                         if isinstance(proto, _sshv1server):
                             # legacy getfiles method which only works over ssh
                             caps.append(constants.NETWORK_CAP_LEGACY_SSH_GETFILES)
                         caps.append(b'x_rfl_getflogheads')
                         caps.append(b'x_rfl_getfile')
                     return caps
                 extensions.wrapfunction(wireprotov1server, b'_capabilities', _capabilities)
                 def _adjustlinkrev(orig, self, *args, **kwargs):
                     # When generating file blobs, taking the real path is too slow on large
                     # repos, so force it to just return the linkrev directly.
                     repo = self._repo
                     if util.safehasattr(repo, b'forcelinkrev') and repo.forcelinkrev:
                         return self._filelog.linkrev(self._filelog.rev(self._filenode))
                     return orig(self, *args, **kwargs)
                 extensions.wrapfunction(
                     context.basefilectx, b'_adjustlinkrev', _adjustlinkrev
                 )
                 def _iscmd(orig, cmd):
                     if cmd == b'x_rfl_getfiles':
                         return False
                     return orig(cmd)
                 extensions.wrapfunction(wireprotoserver, b'iscmd', _iscmd)
             def _loadfileblob(repo, cachepath, path, node):
                 filecachepath = os.path.join(cachepath, path, hex(node))
                 if not os.path.exists(filecachepath) or os.path.getsize(filecachepath) == 0:
                     filectx = repo.filectx(path, fileid=node)
                     if filectx.node() == nullid:
                         repo.changelog = changelog.changelog(repo.svfs)
                         filectx = repo.filectx(path, fileid=node)
                     text = createfileblob(filectx)
                     # TODO configurable compression engines
                     text = zlib.compress(text)
                     # everything should be user & group read/writable
                     oldumask = os.umask(0o002)
                     try:
                         dirname = os.path.dirname(filecachepath)
                         if not os.path.exists(dirname):
                             try:
                                 os.makedirs(dirname)
                             except OSError as ex:
                                 if ex.errno != errno.EEXIST:
                                     raise
                         f = None
                         try:
                             f = util.atomictempfile(filecachepath, b"wb")
                             f.write(text)
                         except (IOError, OSError):
                             # Don't abort if the user only has permission to read,
                             # and not write.
                             pass
                         finally:
                             if f:
                                 f.close()
                     finally:
                         os.umask(oldumask)
                 else:
                     with open(filecachepath, b"rb") as f:
                         text = f.read()
                 return text
             def getflogheads(repo, proto, path):
                 """A server api for requesting a filelog's heads"""
                 flog = repo.file(path)
                 heads = flog.heads()
                 return b'\n'.join((hex(head) for head in heads if head != nullid))
             def getfile(repo, proto, file, node):
                 """A server api for requesting a particular version of a file. Can be used
                 in batches to request many files at once. The return protocol is:
                 <errorcode>\0<data/errormsg> where <errorcode> is 0 for success or
                 non-zero for an error.
                 data is a compressed blob with revlog flag and ancestors information. See
                 createfileblob for its content.
                 """
                 if shallowutil.isenabled(repo):
                     return b'1\0' + _(b'cannot fetch remote files from shallow repo')
                 cachepath = repo.ui.config(b"remotefilelog", b"servercachepath")
                 if not cachepath:
                     cachepath = os.path.join(repo.path, b"remotefilelogcache")
                 node = bin(node.strip())
                 if node == nullid:
                     return b'0\0'
                 return b'0\0' + _loadfileblob(repo, cachepath, file, node)
             def getfiles(repo, proto):
                 """A server api for requesting particular versions of particular files."""
                 if shallowutil.isenabled(repo):
                     raise error.Abort(_(b'cannot fetch remote files from shallow repo'))
                 if not isinstance(proto, _sshv1server):
                     raise error.Abort(_(b'cannot fetch remote files over non-ssh protocol'))
                 def streamer():
                     fin = proto._fin
                     cachepath = repo.ui.config(b"remotefilelog", b"servercachepath")
                     if not cachepath:
                         cachepath = os.path.join(repo.path, b"remotefilelogcache")
                     while True:
                         request = fin.readline()[:-1]
                         if not request:
                             break
                         node = bin(request[:40])
                         if node == nullid:
                             yield b'0\n'
                             continue
                         path = request[40:]
                         text = _loadfileblob(repo, cachepath, path, node)
                         yield b'%d\n%s' % (len(text), text)
                         # it would be better to only flush after processing a whole batch
                         # but currently we don't know if there are more requests coming
                         proto._fout.flush()
                 return wireprototypes.streamres(streamer())
             def createfileblob(filectx):
                 """
                 format:
                     v0:
                         str(len(rawtext)) + '\0' + rawtext + ancestortext
                     v1:
                         'v1' + '\n' + metalist + '\0' + rawtext + ancestortext
                         metalist := metalist + '\n' + meta | meta
                         meta := sizemeta | flagmeta
                         sizemeta := METAKEYSIZE + str(len(rawtext))
                         flagmeta := METAKEYFLAG + str(flag)
                         note: sizemeta must exist. METAKEYFLAG and METAKEYSIZE must have a
                         length of 1.
                 """
                 flog = filectx.filelog()
                 frev = filectx.filerev()
                 revlogflags = flog._revlog.flags(frev)
                 if revlogflags == 0:
                     # normal files
                     text = filectx.data()
                 else:
                     # lfs, read raw revision data
                     text = flog.rawdata(frev)
                 repo = filectx._repo
                 ancestors = [filectx]
                 try:
                     repo.forcelinkrev = True
                     ancestors.extend([f for f in filectx.ancestors()])
                     ancestortext = b""
                     for ancestorctx in ancestors:
                         parents = ancestorctx.parents()
                         p1 = nullid
                         p2 = nullid
                         if len(parents) > 0:
                             p1 = parents[0].filenode()
                         if len(parents) > 1:
                             p2 = parents[1].filenode()
                         copyname = b""
                         rename = ancestorctx.renamed()
                         if rename:
                             copyname = rename[0]
                         linknode = ancestorctx.node()
                         ancestortext += b"%s%s%s%s%s\0" % (
                             ancestorctx.filenode(),
                             p1,
                             p2,
                             linknode,
                             copyname,
                         )
                 finally:
                     repo.forcelinkrev = False
                 header = shallowutil.buildfileblobheader(len(text), revlogflags)
                 return b"%s\0%s%s" % (header, text, ancestortext)
             def gcserver(ui, repo):
                 if not repo.ui.configbool(b"remotefilelog", b"server"):
                     return
                 neededfiles = set()
                 heads = repo.revs(b"heads(tip~25000:) - null")
                 cachepath = repo.vfs.join(b"remotefilelogcache")
                 for head in heads:
                     mf = repo[head].manifest()
                     for filename, filenode in pycompat.iteritems(mf):
                         filecachepath = os.path.join(cachepath, filename, hex(filenode))
                         neededfiles.add(filecachepath)
                 # delete unneeded older files
                 days = repo.ui.configint(b"remotefilelog", b"serverexpiration")
                 expiration = time.time() - (days * 24 * 60 * 60)
                 progress = ui.makeprogress(_(b"removing old server cache"), unit=b"files")
                 progress.update(0)
                 for root, dirs, files in os.walk(cachepath):
                     for file in files:
                         filepath = os.path.join(root, file)
                         progress.increment()
                         if filepath in neededfiles:
                             continue
                         stat = os.stat(filepath)
                         if stat.st_mtime < expiration:
                             os.remove(filepath)
                 progress.complete()

mercurial/repair.py

0 +1 -1

             # repair.py - functions for repository repair for mercurial
             #
             # Copyright 2005, 2006 Chris Mason <mason@suse.com>
             # Copyright 2007 Olivia Mackall
             #
             # This software may be used and distributed according to the terms of the
             # GNU General Public License version 2 or any later version.
             from __future__ import absolute_import
             import errno
             from .i18n import _
             from .node import (
                 hex,
                 short,
             )
             from . import (
                 bundle2,
                 changegroup,
                 discovery,
                 error,
                 exchange,
                 obsolete,
                 obsutil,
                 pathutil,
                 phases,
                 pycompat,
                 requirements,
                 scmutil,
                 util,
             )
             from .utils import (
                 hashutil,
                 stringutil,
             )
             def backupbundle(
                 repo, bases, heads, node, suffix, compress=True, obsolescence=True
             ):
                 """create a bundle with the specified revisions as a backup"""
                 backupdir = b"strip-backup"
                 vfs = repo.vfs
                 if not vfs.isdir(backupdir):
                     vfs.mkdir(backupdir)
                 # Include a hash of all the nodes in the filename for uniqueness
                 allcommits = repo.set(b'%ln::%ln', bases, heads)
                 allhashes = sorted(c.hex() for c in allcommits)
                 totalhash = hashutil.sha1(b''.join(allhashes)).digest()
                 name = b"%s/%s-%s-%s.hg" % (
                     backupdir,
                     short(node),
                     hex(totalhash[:4]),
                     suffix,
                 )
                 cgversion = changegroup.localversion(repo)
                 comp = None
                 if cgversion != b'01':
                     bundletype = b"HG20"
                     if compress:
                         comp = b'BZ'
                 elif compress:
                     bundletype = b"HG10BZ"
                 else:
                     bundletype = b"HG10UN"
                 outgoing = discovery.outgoing(repo, missingroots=bases, ancestorsof=heads)
                 contentopts = {
                     b'cg.version': cgversion,
                     b'obsolescence': obsolescence,
                     b'phases': True,
                 }
                 return bundle2.writenewbundle(
                     repo.ui,
                     repo,
                     b'strip',
                     name,
                     bundletype,
                     outgoing,
                     contentopts,
                     vfs,
                     compression=comp,
                 )
             def _collectfiles(repo, striprev):
                 """find out the filelogs affected by the strip"""
                 files = set()
                 for x in pycompat.xrange(striprev, len(repo)):
                     files.update(repo[x].files())
                 return sorted(files)
             def _collectrevlog(revlog, striprev):
                 _, brokenset = revlog.getstrippoint(striprev)
                 return [revlog.linkrev(r) for r in brokenset]
             def _collectbrokencsets(repo, files, striprev):
                 """return the changesets which will be broken by the truncation"""
                 s = set()
                 for revlog in manifestrevlogs(repo):
                     s.update(_collectrevlog(revlog, striprev))
                 for fname in files:
                     s.update(_collectrevlog(repo.file(fname), striprev))
                 return s
             def strip(ui, repo, nodelist, backup=True, topic=b'backup'):
                 # This function requires the caller to lock the repo, but it operates
                 # within a transaction of its own, and thus requires there to be no current
                 # transaction when it is called.
                 if repo.currenttransaction() is not None:
                     raise error.ProgrammingError(b'cannot strip from inside a transaction')
                 # Simple way to maintain backwards compatibility for this
                 # argument.
                 if backup in [b'none', b'strip']:
                     backup = False
                 repo = repo.unfiltered()
                 repo.destroying()
                 vfs = repo.vfs
                 # load bookmark before changelog to avoid side effect from outdated
                 # changelog (see repo._refreshchangelog)
                 repo._bookmarks
                 cl = repo.changelog
                 # TODO handle undo of merge sets
                 if isinstance(nodelist, bytes):
                     nodelist = [nodelist]
                 striplist = [cl.rev(node) for node in nodelist]
                 striprev = min(striplist)
                 files = _collectfiles(repo, striprev)
                 saverevs = _collectbrokencsets(repo, files, striprev)
                 # Some revisions with rev > striprev may not be descendants of striprev.
                 # We have to find these revisions and put them in a bundle, so that
                 # we can restore them after the truncations.
                 # To create the bundle we use repo.changegroupsubset which requires
                 # the list of heads and bases of the set of interesting revisions.
                 # (head = revision in the set that has no descendant in the set;
                 #  base = revision in the set that has no ancestor in the set)
                 tostrip = set(striplist)
                 saveheads = set(saverevs)
                 for r in cl.revs(start=striprev + 1):
                     if any(p in tostrip for p in cl.parentrevs(r)):
                         tostrip.add(r)
                     if r not in tostrip:
                         saverevs.add(r)
                         saveheads.difference_update(cl.parentrevs(r))
                         saveheads.add(r)
                 saveheads = [cl.node(r) for r in saveheads]
                 # compute base nodes
                 if saverevs:
                     descendants = set(cl.descendants(saverevs))
                     saverevs.difference_update(descendants)
                 savebases = [cl.node(r) for r in saverevs]
                 stripbases = [cl.node(r) for r in tostrip]
                 stripobsidx = obsmarkers = ()
                 if repo.ui.configbool(b'devel', b'strip-obsmarkers'):
                     obsmarkers = obsutil.exclusivemarkers(repo, stripbases)
                 if obsmarkers:
                     stripobsidx = [
                         i for i, m in enumerate(repo.obsstore) if m in obsmarkers
                     ]
                 newbmtarget, updatebm = _bookmarkmovements(repo, tostrip)
                 backupfile = None
                 node = nodelist[-1]
                 if backup:
                     backupfile = _createstripbackup(repo, stripbases, node, topic)
                 # create a changegroup for all the branches we need to keep
                 tmpbundlefile = None
                 if saveheads:
                     # do not compress temporary bundle if we remove it from disk later
                     #
                     # We do not include obsolescence, it might re-introduce prune markers
                     # we are trying to strip.  This is harmless since the stripped markers
                     # are already backed up and we did not touched the markers for the
                     # saved changesets.
                     tmpbundlefile = backupbundle(
                         repo,
                         savebases,
                         saveheads,
                         node,
                         b'temp',
                         compress=False,
                         obsolescence=False,
                     )
                 with ui.uninterruptible():
                     try:
                         with repo.transaction(b"strip") as tr:
                             # TODO this code violates the interface abstraction of the
                             # transaction and makes assumptions that file storage is
                             # using append-only files. We'll need some kind of storage
                             # API to handle stripping for us.
                             oldfiles = set(tr._offsetmap.keys())
                             oldfiles.update(tr._newfiles)
                             tr.startgroup()
                             cl.strip(striprev, tr)
                             stripmanifest(repo, striprev, tr, files)
                             for fn in files:
                                 repo.file(fn).strip(striprev, tr)
                             tr.endgroup()
                             entries = tr.readjournal()
                             for file, troffset in entries:
                                 if file in oldfiles:
                                     continue
                                 with repo.svfs(file, b'a', checkambig=True) as fp:
                                     fp.truncate(troffset)
                                 if troffset == 0:
                                     repo.store.markremoved(file)
                             deleteobsmarkers(repo.obsstore, stripobsidx)
                             del repo.obsstore
                             repo.invalidatevolatilesets()
                             repo._phasecache.filterunknown(repo)
                         if tmpbundlefile:
                             ui.note(_(b"adding branch\n"))
                             f = vfs.open(tmpbundlefile, b"rb")
                             gen = exchange.readbundle(ui, f, tmpbundlefile, vfs)
                             if not repo.ui.verbose:
                                 # silence internal shuffling chatter
                                 repo.ui.pushbuffer()
                             tmpbundleurl = b'bundle:' + vfs.join(tmpbundlefile)
                             txnname = b'strip'
                             if not isinstance(gen, bundle2.unbundle20):
                                 txnname = b"strip\n%s" % util.hidepassword(tmpbundleurl)
                             with repo.transaction(txnname) as tr:
                                 bundle2.applybundle(
                                     repo, gen, tr, source=b'strip', url=tmpbundleurl
                                 )
                             if not repo.ui.verbose:
                                 repo.ui.popbuffer()
                             f.close()
                         with repo.transaction(b'repair') as tr:
                             bmchanges = [(m, repo[newbmtarget].node()) for m in updatebm]
                             repo._bookmarks.applychanges(repo, tr, bmchanges)
                         # remove undo files
                         for undovfs, undofile in repo.undofiles():
                             try:
                                 undovfs.unlink(undofile)
                             except OSError as e:
                                 if e.errno != errno.ENOENT:
                                     ui.warn(
                                         _(b'error removing %s: %s\n')
                                         % (
                                             undovfs.join(undofile),
                                             stringutil.forcebytestr(e),
                                         )
                                     )
                     except:  # re-raises
                         if backupfile:
                             ui.warn(
                                 _(b"strip failed, backup bundle stored in '%s'\n")
                                 % vfs.join(backupfile)
                             )
                         if tmpbundlefile:
                             ui.warn(
                                 _(b"strip failed, unrecovered changes stored in '%s'\n")
                                 % vfs.join(tmpbundlefile)
                             )
                             ui.warn(
                                 _(
                                     b"(fix the problem, then recover the changesets with "
                                     b"\"hg unbundle '%s'\")\n"
                                 )
                                 % vfs.join(tmpbundlefile)
                             )
                         raise
                     else:
                         if tmpbundlefile:
                             # Remove temporary bundle only if there were no exceptions
                             vfs.unlink(tmpbundlefile)
                 repo.destroyed()
                 # return the backup file path (or None if 'backup' was False) so
                 # extensions can use it
                 return backupfile
             def softstrip(ui, repo, nodelist, backup=True, topic=b'backup'):
                 """perform a "soft" strip using the archived phase"""
                 tostrip = [c.node() for c in repo.set(b'sort(%ln::)', nodelist)]
                 if not tostrip:
                     return None
                 backupfile = None
                 if backup:
                     node = tostrip[0]
                     backupfile = _createstripbackup(repo, tostrip, node, topic)
                 newbmtarget, updatebm = _bookmarkmovements(repo, tostrip)
                 with repo.transaction(b'strip') as tr:
                     phases.retractboundary(repo, tr, phases.archived, tostrip)
                     bmchanges = [(m, repo[newbmtarget].node()) for m in updatebm]
                     repo._bookmarks.applychanges(repo, tr, bmchanges)
                 return backupfile
             def _bookmarkmovements(repo, tostrip):
                 # compute necessary bookmark movement
                 bm = repo._bookmarks
                 updatebm = []
                 for m in bm:
                     rev = repo[bm[m]].rev()
                     if rev in tostrip:
                         updatebm.append(m)
                 newbmtarget = None
                 # If we need to move bookmarks, compute bookmark
                 # targets. Otherwise we can skip doing this logic.
                 if updatebm:
                     # For a set s, max(parents(s) - s) is the same as max(heads(::s - s)),
                     # but is much faster
                     newbmtarget = repo.revs(b'max(parents(%ld) - (%ld))', tostrip, tostrip)
                     if newbmtarget:
                         newbmtarget = repo[newbmtarget.first()].node()
                     else:
                         newbmtarget = b'.'
                 return newbmtarget, updatebm
             def _createstripbackup(repo, stripbases, node, topic):
                 # backup the changeset we are about to strip
                 vfs = repo.vfs
                 cl = repo.changelog
                 backupfile = backupbundle(repo, stripbases, cl.heads(), node, topic)
                 repo.ui.status(_(b"saved backup bundle to %s\n") % vfs.join(backupfile))
                 repo.ui.log(
                     b"backupbundle", b"saved backup bundle to %s\n", vfs.join(backupfile)
                 )
                 return backupfile
             def safestriproots(ui, repo, nodes):
                 """return list of roots of nodes where descendants are covered by nodes"""
                 torev = repo.unfiltered().changelog.rev
                 revs = {torev(n) for n in nodes}
                 # tostrip = wanted - unsafe = wanted - ancestors(orphaned)
                 # orphaned = affected - wanted
                 # affected = descendants(roots(wanted))
                 # wanted = revs
                 revset = b'%ld - ( ::( (roots(%ld):: and not _phase(%s)) -%ld) )'
                 tostrip = set(repo.revs(revset, revs, revs, phases.internal, revs))
                 notstrip = revs - tostrip
                 if notstrip:
                     nodestr = b', '.join(sorted(short(repo[n].node()) for n in notstrip))
                     ui.warn(
                         _(b'warning: orphaned descendants detected, not stripping %s\n')
                         % nodestr
                     )
                 return [c.node() for c in repo.set(b'roots(%ld)', tostrip)]
             class stripcallback(object):
                 """used as a transaction postclose callback"""
                 def __init__(self, ui, repo, backup, topic):
                     self.ui = ui
                     self.repo = repo
                     self.backup = backup
                     self.topic = topic or b'backup'
                     self.nodelist = []
                 def addnodes(self, nodes):
                     self.nodelist.extend(nodes)
                 def __call__(self, tr):
                     roots = safestriproots(self.ui, self.repo, self.nodelist)
                     if roots:
                         strip(self.ui, self.repo, roots, self.backup, self.topic)
             def delayedstrip(ui, repo, nodelist, topic=None, backup=True):
                 """like strip, but works inside transaction and won't strip irreverent revs
                 nodelist must explicitly contain all descendants. Otherwise a warning will
                 be printed that some nodes are not stripped.
                 Will do a backup if `backup` is True. The last non-None "topic" will be
                 used as the backup topic name. The default backup topic name is "backup".
                 """
                 tr = repo.currenttransaction()
                 if not tr:
                     nodes = safestriproots(ui, repo, nodelist)
                     return strip(ui, repo, nodes, backup=backup, topic=topic)
                 # transaction postclose callbacks are called in alphabet order.
                 # use '\xff' as prefix so we are likely to be called last.
                 callback = tr.getpostclose(b'\xffstrip')
                 if callback is None:
                     callback = stripcallback(ui, repo, backup=backup, topic=topic)
                     tr.addpostclose(b'\xffstrip', callback)
                 if topic:
                     callback.topic = topic
                 callback.addnodes(nodelist)
             def stripmanifest(repo, striprev, tr, files):
                 for revlog in manifestrevlogs(repo):
                     revlog.strip(striprev, tr)
             def manifestrevlogs(repo):
                 yield repo.manifestlog.getstorage(b'')
                 if scmutil.istreemanifest(repo):
                     # This logic is safe if treemanifest isn't enabled, but also
                     # pointless, so we skip it if treemanifest isn't enabled.
-                    for unencoded, encoded, size in repo.store.datafiles():
+                    for t, unencoded, encoded, size in repo.store.datafiles():
                         if unencoded.startswith(b'meta/') and unencoded.endswith(
                             b'00manifest.i'
                         ):
                             dir = unencoded[5:-12]
                             yield repo.manifestlog.getstorage(dir)
             def rebuildfncache(ui, repo):
                 """Rebuilds the fncache file from repo history.
                 Missing entries will be added. Extra entries will be removed.
                 """
                 repo = repo.unfiltered()
                 if requirements.FNCACHE_REQUIREMENT not in repo.requirements:
                     ui.warn(
                         _(
                             b'(not rebuilding fncache because repository does not '
                             b'support fncache)\n'
                         )
                     )
                     return
                 with repo.lock():
                     fnc = repo.store.fncache
                     fnc.ensureloaded(warn=ui.warn)
                     oldentries = set(fnc.entries)
                     newentries = set()
                     seenfiles = set()
                     progress = ui.makeprogress(
                         _(b'rebuilding'), unit=_(b'changesets'), total=len(repo)
                     )
                     for rev in repo:
                         progress.update(rev)
                         ctx = repo[rev]
                         for f in ctx.files():
                             # This is to minimize I/O.
                             if f in seenfiles:
                                 continue
                             seenfiles.add(f)
                             i = b'data/%s.i' % f
                             d = b'data/%s.d' % f
                             if repo.store._exists(i):
                                 newentries.add(i)
                             if repo.store._exists(d):
                                 newentries.add(d)
                     progress.complete()
                     if requirements.TREEMANIFEST_REQUIREMENT in repo.requirements:
                         # This logic is safe if treemanifest isn't enabled, but also
                         # pointless, so we skip it if treemanifest isn't enabled.
                         for dir in pathutil.dirs(seenfiles):
                             i = b'meta/%s/00manifest.i' % dir
                             d = b'meta/%s/00manifest.d' % dir
                             if repo.store._exists(i):
                                 newentries.add(i)
                             if repo.store._exists(d):
                                 newentries.add(d)
                     addcount = len(newentries - oldentries)
                     removecount = len(oldentries - newentries)
                     for p in sorted(oldentries - newentries):
                         ui.write(_(b'removing %s\n') % p)
                     for p in sorted(newentries - oldentries):
                         ui.write(_(b'adding %s\n') % p)
                     if addcount or removecount:
                         ui.write(
                             _(b'%d items added, %d removed from fncache\n')
                             % (addcount, removecount)
                         )
                         fnc.entries = newentries
                         fnc._dirty = True
                         with repo.transaction(b'fncache') as tr:
                             fnc.write(tr)
                     else:
                         ui.write(_(b'fncache already up to date\n'))
             def deleteobsmarkers(obsstore, indices):
                 """Delete some obsmarkers from obsstore and return how many were deleted
                 'indices' is a list of ints which are the indices
                 of the markers to be deleted.
                 Every invocation of this function completely rewrites the obsstore file,
                 skipping the markers we want to be removed. The new temporary file is
                 created, remaining markers are written there and on .close() this file
                 gets atomically renamed to obsstore, thus guaranteeing consistency."""
                 if not indices:
                     # we don't want to rewrite the obsstore with the same content
                     return
                 left = []
                 current = obsstore._all
                 n = 0
                 for i, m in enumerate(current):
                     if i in indices:
                         n += 1
                         continue
                     left.append(m)
                 newobsstorefile = obsstore.svfs(b'obsstore', b'w', atomictemp=True)
                 for bytes in obsolete.encodemarkers(left, True, obsstore._version):
                     newobsstorefile.write(bytes)
                 newobsstorefile.close()
                 return n

mercurial/store.py

0 +56 -13

             # store.py - repository store handling for Mercurial
             #
             # Copyright 2008 Olivia Mackall <olivia@selenic.com>
             #
             # This software may be used and distributed according to the terms of the
             # GNU General Public License version 2 or any later version.
             from __future__ import absolute_import
             import errno
             import functools
             import os
             import stat
             from .i18n import _
             from .pycompat import getattr
             from .node import hex
             from . import (
                 changelog,
                 error,
                 manifest,
                 policy,
                 pycompat,
                 util,
                 vfs as vfsmod,
             )
             from .utils import hashutil
             parsers = policy.importmod('parsers')
             # how much bytes should be read from fncache in one read
             # It is done to prevent loading large fncache files into memory
             fncache_chunksize = 10 ** 6
             def _matchtrackedpath(path, matcher):
                 """parses a fncache entry and returns whether the entry is tracking a path
                 matched by matcher or not.
                 If matcher is None, returns True"""
                 if matcher is None:
                     return True
                 path = decodedir(path)
                 if path.startswith(b'data/'):
                     return matcher(path[len(b'data/') : -len(b'.i')])
                 elif path.startswith(b'meta/'):
                     return matcher.visitdir(path[len(b'meta/') : -len(b'/00manifest.i')])
                 raise error.ProgrammingError(b"cannot decode path %s" % path)
             # This avoids a collision between a file named foo and a dir named
             # foo.i or foo.d
             def _encodedir(path):
                 """
                 >>> _encodedir(b'data/foo.i')
                 'data/foo.i'
                 >>> _encodedir(b'data/foo.i/bla.i')
                 'data/foo.i.hg/bla.i'
                 >>> _encodedir(b'data/foo.i.hg/bla.i')
                 'data/foo.i.hg.hg/bla.i'
                 >>> _encodedir(b'data/foo.i\\ndata/foo.i/bla.i\\ndata/foo.i.hg/bla.i\\n')
                 'data/foo.i\\ndata/foo.i.hg/bla.i\\ndata/foo.i.hg.hg/bla.i\\n'
                 """
                 return (
                     path.replace(b".hg/", b".hg.hg/")
                     .replace(b".i/", b".i.hg/")
                     .replace(b".d/", b".d.hg/")
                 )
             encodedir = getattr(parsers, 'encodedir', _encodedir)
             def decodedir(path):
                 """
                 >>> decodedir(b'data/foo.i')
                 'data/foo.i'
                 >>> decodedir(b'data/foo.i.hg/bla.i')
                 'data/foo.i/bla.i'
                 >>> decodedir(b'data/foo.i.hg.hg/bla.i')
                 'data/foo.i.hg/bla.i'
                 """
                 if b".hg/" not in path:
                     return path
                 return (
                     path.replace(b".d.hg/", b".d/")
                     .replace(b".i.hg/", b".i/")
                     .replace(b".hg.hg/", b".hg/")
                 )
             def _reserved():
                 """characters that are problematic for filesystems
                 * ascii escapes (0..31)
                 * ascii hi (126..255)
                 * windows specials
                 these characters will be escaped by encodefunctions
                 """
                 winreserved = [ord(x) for x in u'\\:*?"<>|']
                 for x in range(32):
                     yield x
                 for x in range(126, 256):
                     yield x
                 for x in winreserved:
                     yield x
             def _buildencodefun():
                 """
                 >>> enc, dec = _buildencodefun()
                 >>> enc(b'nothing/special.txt')
                 'nothing/special.txt'
                 >>> dec(b'nothing/special.txt')
                 'nothing/special.txt'
                 >>> enc(b'HELLO')
                 '_h_e_l_l_o'
                 >>> dec(b'_h_e_l_l_o')
                 'HELLO'
                 >>> enc(b'hello:world?')
                 'hello~3aworld~3f'
                 >>> dec(b'hello~3aworld~3f')
                 'hello:world?'
                 >>> enc(b'the\\x07quick\\xADshot')
                 'the~07quick~adshot'
                 >>> dec(b'the~07quick~adshot')
                 'the\\x07quick\\xadshot'
                 """
                 e = b'_'
                 xchr = pycompat.bytechr
                 asciistr = list(map(xchr, range(127)))
                 capitals = list(range(ord(b"A"), ord(b"Z") + 1))
                 cmap = {x: x for x in asciistr}
                 for x in _reserved():
                     cmap[xchr(x)] = b"~%02x" % x
                 for x in capitals + [ord(e)]:
                     cmap[xchr(x)] = e + xchr(x).lower()
                 dmap = {}
                 for k, v in pycompat.iteritems(cmap):
                     dmap[v] = k
                 def decode(s):
                     i = 0
                     while i < len(s):
                         for l in pycompat.xrange(1, 4):
                             try:
                                 yield dmap[s[i : i + l]]
                                 i += l
                                 break
                             except KeyError:
                                 pass
                         else:
                             raise KeyError
                 return (
                     lambda s: b''.join(
                         [cmap[s[c : c + 1]] for c in pycompat.xrange(len(s))]
                     ),
                     lambda s: b''.join(list(decode(s))),
                 )
             _encodefname, _decodefname = _buildencodefun()
             def encodefilename(s):
                 """
                 >>> encodefilename(b'foo.i/bar.d/bla.hg/hi:world?/HELLO')
                 'foo.i.hg/bar.d.hg/bla.hg.hg/hi~3aworld~3f/_h_e_l_l_o'
                 """
                 return _encodefname(encodedir(s))
             def decodefilename(s):
                 """
                 >>> decodefilename(b'foo.i.hg/bar.d.hg/bla.hg.hg/hi~3aworld~3f/_h_e_l_l_o')
                 'foo.i/bar.d/bla.hg/hi:world?/HELLO'
                 """
                 return decodedir(_decodefname(s))
             def _buildlowerencodefun():
                 """
                 >>> f = _buildlowerencodefun()
                 >>> f(b'nothing/special.txt')
                 'nothing/special.txt'
                 >>> f(b'HELLO')
                 'hello'
                 >>> f(b'hello:world?')
                 'hello~3aworld~3f'
                 >>> f(b'the\\x07quick\\xADshot')
                 'the~07quick~adshot'
                 """
                 xchr = pycompat.bytechr
                 cmap = {xchr(x): xchr(x) for x in pycompat.xrange(127)}
                 for x in _reserved():
                     cmap[xchr(x)] = b"~%02x" % x
                 for x in range(ord(b"A"), ord(b"Z") + 1):
                     cmap[xchr(x)] = xchr(x).lower()
                 def lowerencode(s):
                     return b"".join([cmap[c] for c in pycompat.iterbytestr(s)])
                 return lowerencode
             lowerencode = getattr(parsers, 'lowerencode', None) or _buildlowerencodefun()
             # Windows reserved names: con, prn, aux, nul, com1..com9, lpt1..lpt9
             _winres3 = (b'aux', b'con', b'prn', b'nul')  # length 3
             _winres4 = (b'com', b'lpt')  # length 4 (with trailing 1..9)
             def _auxencode(path, dotencode):
                 """
                 Encodes filenames containing names reserved by Windows or which end in
                 period or space. Does not touch other single reserved characters c.
                 Specifically, c in '\\:*?"<>|' or ord(c) <= 31 are *not* encoded here.
                 Additionally encodes space or period at the beginning, if dotencode is
                 True. Parameter path is assumed to be all lowercase.
                 A segment only needs encoding if a reserved name appears as a
                 basename (e.g. "aux", "aux.foo"). A directory or file named "foo.aux"
                 doesn't need encoding.
                 >>> s = b'.foo/aux.txt/txt.aux/con/prn/nul/foo.'
                 >>> _auxencode(s.split(b'/'), True)
                 ['~2efoo', 'au~78.txt', 'txt.aux', 'co~6e', 'pr~6e', 'nu~6c', 'foo~2e']
                 >>> s = b'.com1com2/lpt9.lpt4.lpt1/conprn/com0/lpt0/foo.'
                 >>> _auxencode(s.split(b'/'), False)
                 ['.com1com2', 'lp~749.lpt4.lpt1', 'conprn', 'com0', 'lpt0', 'foo~2e']
                 >>> _auxencode([b'foo. '], True)
                 ['foo.~20']
                 >>> _auxencode([b' .foo'], True)
                 ['~20.foo']
                 """
                 for i, n in enumerate(path):
                     if not n:
                         continue
                     if dotencode and n[0] in b'. ':
                         n = b"~%02x" % ord(n[0:1]) + n[1:]
                         path[i] = n
                     else:
                         l = n.find(b'.')
                         if l == -1:
                             l = len(n)
                         if (l == 3 and n[:3] in _winres3) or (
                             l == 4
                             and n[3:4] <= b'9'
                             and n[3:4] >= b'1'
                             and n[:3] in _winres4
                         ):
                             # encode third letter ('aux' -> 'au~78')
                             ec = b"~%02x" % ord(n[2:3])
                             n = n[0:2] + ec + n[3:]
                             path[i] = n
                     if n[-1] in b'. ':
                         # encode last period or space ('foo...' -> 'foo..~2e')
                         path[i] = n[:-1] + b"~%02x" % ord(n[-1:])
                 return path
             _maxstorepathlen = 120
             _dirprefixlen = 8
             _maxshortdirslen = 8 * (_dirprefixlen + 1) - 4
             def _hashencode(path, dotencode):
                 digest = hex(hashutil.sha1(path).digest())
                 le = lowerencode(path[5:]).split(b'/')  # skips prefix 'data/' or 'meta/'
                 parts = _auxencode(le, dotencode)
                 basename = parts[-1]
                 _root, ext = os.path.splitext(basename)
                 sdirs = []
                 sdirslen = 0
                 for p in parts[:-1]:
                     d = p[:_dirprefixlen]
                     if d[-1] in b'. ':
                         # Windows can't access dirs ending in period or space
                         d = d[:-1] + b'_'
                     if sdirslen == 0:
                         t = len(d)
                     else:
                         t = sdirslen + 1 + len(d)
                         if t > _maxshortdirslen:
                             break
                     sdirs.append(d)
                     sdirslen = t
                 dirs = b'/'.join(sdirs)
                 if len(dirs) > 0:
                     dirs += b'/'
                 res = b'dh/' + dirs + digest + ext
                 spaceleft = _maxstorepathlen - len(res)
                 if spaceleft > 0:
                     filler = basename[:spaceleft]
                     res = b'dh/' + dirs + filler + digest + ext
                 return res
             def _hybridencode(path, dotencode):
                 """encodes path with a length limit
                 Encodes all paths that begin with 'data/', according to the following.
                 Default encoding (reversible):
                 Encodes all uppercase letters 'X' as '_x'. All reserved or illegal
                 characters are encoded as '~xx', where xx is the two digit hex code
                 of the character (see encodefilename).
                 Relevant path components consisting of Windows reserved filenames are
                 masked by encoding the third character ('aux' -> 'au~78', see _auxencode).
                 Hashed encoding (not reversible):
                 If the default-encoded path is longer than _maxstorepathlen, a
                 non-reversible hybrid hashing of the path is done instead.
                 This encoding uses up to _dirprefixlen characters of all directory
                 levels of the lowerencoded path, but not more levels than can fit into
                 _maxshortdirslen.
                 Then follows the filler followed by the sha digest of the full path.
                 The filler is the beginning of the basename of the lowerencoded path
                 (the basename is everything after the last path separator). The filler
                 is as long as possible, filling in characters from the basename until
                 the encoded path has _maxstorepathlen characters (or all chars of the
                 basename have been taken).
                 The extension (e.g. '.i' or '.d') is preserved.
                 The string 'data/' at the beginning is replaced with 'dh/', if the hashed
                 encoding was used.
                 """
                 path = encodedir(path)
                 ef = _encodefname(path).split(b'/')
                 res = b'/'.join(_auxencode(ef, dotencode))
                 if len(res) > _maxstorepathlen:
                     res = _hashencode(path, dotencode)
                 return res
             def _pathencode(path):
                 de = encodedir(path)
                 if len(path) > _maxstorepathlen:
                     return _hashencode(de, True)
                 ef = _encodefname(de).split(b'/')
                 res = b'/'.join(_auxencode(ef, True))
                 if len(res) > _maxstorepathlen:
                     return _hashencode(de, True)
                 return res
             _pathencode = getattr(parsers, 'pathencode', _pathencode)
             def _plainhybridencode(f):
                 return _hybridencode(f, False)
             def _calcmode(vfs):
                 try:
                     # files in .hg/ will be created using this mode
                     mode = vfs.stat().st_mode
                     # avoid some useless chmods
                     if (0o777 & ~util.umask) == (0o777 & mode):
                         mode = None
                 except OSError:
                     mode = None
                 return mode
             _data = [
                 b'bookmarks',
                 b'narrowspec',
                 b'data',
                 b'meta',
                 b'00manifest.d',
                 b'00manifest.i',
                 b'00changelog.d',
                 b'00changelog.i',
                 b'phaseroots',
                 b'obsstore',
                 b'requires',
             ]
-            REVLOG_FILES_EXT = (b'.i', b'.d', b'.n', b'.nd')
+            REVLOG_FILES_MAIN_EXT = (b'.i', b'i.tmpcensored')
+            REVLOG_FILES_OTHER_EXT = (b'.d', b'.n', b'.nd', b'd.tmpcensored')
+            def is_revlog(f, kind, st):
+                if kind != stat.S_IFREG:
+                    return None
+                return revlog_type(f)
+            def revlog_type(f):
+                if f.endswith(REVLOG_FILES_MAIN_EXT):
+                    return FILEFLAGS_REVLOG_MAIN
+                elif f.endswith(REVLOG_FILES_OTHER_EXT):
+                    return FILETYPE_FILELOG_OTHER
-            def isrevlog(f, kind, st):
+            # the file is part of changelog data
-                if kind != stat.S_IFREG:
+            FILEFLAGS_CHANGELOG = 1 << 13
-                    return False
+            # the file is part of manifest data
-                return f.endswith(REVLOG_FILES_EXT)
+            FILEFLAGS_MANIFESTLOG = 1 << 12
+            # the file is part of filelog data
+            FILEFLAGS_FILELOG = 1 << 11
+            # file that are not directly part of a revlog
+            FILEFLAGS_OTHER = 1 << 10
+            # the main entry point for a revlog
+            FILEFLAGS_REVLOG_MAIN = 1 << 1
+            # a secondary file for a revlog
+            FILEFLAGS_REVLOG_OTHER = 1 << 0
+            FILETYPE_CHANGELOG_MAIN = FILEFLAGS_CHANGELOG | FILEFLAGS_REVLOG_MAIN
+            FILETYPE_CHANGELOG_OTHER = FILEFLAGS_CHANGELOG | FILEFLAGS_REVLOG_OTHER
+            FILETYPE_MANIFESTLOG_MAIN = FILEFLAGS_MANIFESTLOG | FILEFLAGS_REVLOG_MAIN
+            FILETYPE_MANIFESTLOG_OTHER = FILEFLAGS_MANIFESTLOG | FILEFLAGS_REVLOG_OTHER
+            FILETYPE_FILELOG_MAIN = FILEFLAGS_FILELOG | FILEFLAGS_REVLOG_MAIN
+            FILETYPE_FILELOG_OTHER = FILEFLAGS_FILELOG | FILEFLAGS_REVLOG_OTHER
+            FILETYPE_OTHER = FILEFLAGS_OTHER
             class basicstore(object):
                 '''base class for local repository stores'''
                 def __init__(self, path, vfstype):
                     vfs = vfstype(path)
                     self.path = vfs.base
                     self.createmode = _calcmode(vfs)
                     vfs.createmode = self.createmode
                     self.rawvfs = vfs
                     self.vfs = vfsmod.filtervfs(vfs, encodedir)
                     self.opener = self.vfs
                 def join(self, f):
                     return self.path + b'/' + encodedir(f)
                 def _walk(self, relpath, recurse):
                     '''yields (unencoded, encoded, size)'''
                     path = self.path
                     if relpath:
                         path += b'/' + relpath
                     striplen = len(self.path) + 1
                     l = []
                     if self.rawvfs.isdir(path):
                         visit = [path]
                         readdir = self.rawvfs.readdir
                         while visit:
                             p = visit.pop()
                             for f, kind, st in readdir(p, stat=True):
                                 fp = p + b'/' + f
-                                if isrevlog(f, kind, st):
+                                rl_type = is_revlog(f, kind, st)
+                                if rl_type is not None:
                                     n = util.pconvert(fp[striplen:])
-                                    l.append((decodedir(n), n, st.st_size))
+                                    l.append((rl_type, decodedir(n), n, st.st_size))
                                 elif kind == stat.S_IFDIR and recurse:
                                     visit.append(fp)
                     l.sort()
                     return l
                 def changelog(self, trypending, concurrencychecker=None):
                     return changelog.changelog(
                         self.vfs,
                         trypending=trypending,
                         concurrencychecker=concurrencychecker,
                     )
                 def manifestlog(self, repo, storenarrowmatch):
                     rootstore = manifest.manifestrevlog(repo.nodeconstants, self.vfs)
                     return manifest.manifestlog(self.vfs, repo, rootstore, storenarrowmatch)
                 def datafiles(self, matcher=None):
-                    return self._walk(b'data', True) + self._walk(b'meta', True)
+                    files = self._walk(b'data', True) + self._walk(b'meta', True)
+                    for (t, u, e, s) in files:
+                        yield (FILEFLAGS_FILELOG | t, u, e, s)
                 def topfiles(self):
                     # yield manifest before changelog
-                    return reversed(self._walk(b'', False))
+                    files = reversed(self._walk(b'', False))
+                    for (t, u, e, s) in files:
+                        if u.startswith(b'00changelog'):
+                            yield (FILEFLAGS_CHANGELOG | t, u, e, s)
+                        elif u.startswith(b'00manifest'):
+                            yield (FILEFLAGS_MANIFESTLOG | t, u, e, s)
+                        else:
+                            yield (FILETYPE_OTHER | t, u, e, s)
                 def walk(self, matcher=None):
                     """return file related to data storage (ie: revlogs)
-                    yields (unencoded, encoded, size)
+                    yields (file_type, unencoded, encoded, size)
                     if a matcher is passed, storage files of only those tracked paths
                     are passed with matches the matcher
                     """
                     # yield data files first
                     for x in self.datafiles(matcher):
                         yield x
                     for x in self.topfiles():
                         yield x
                 def copylist(self):
                     return _data
                 def write(self, tr):
                     pass
                 def invalidatecaches(self):
                     pass
                 def markremoved(self, fn):
                     pass
                 def __contains__(self, path):
                     '''Checks if the store contains path'''
                     path = b"/".join((b"data", path))
                     # file?
                     if self.vfs.exists(path + b".i"):
                         return True
                     # dir?
                     if not path.endswith(b"/"):
                         path = path + b"/"
                     return self.vfs.exists(path)
             class encodedstore(basicstore):
                 def __init__(self, path, vfstype):
                     vfs = vfstype(path + b'/store')
                     self.path = vfs.base
                     self.createmode = _calcmode(vfs)
                     vfs.createmode = self.createmode
                     self.rawvfs = vfs
                     self.vfs = vfsmod.filtervfs(vfs, encodefilename)
                     self.opener = self.vfs
                 def datafiles(self, matcher=None):
-                    for a, b, size in super(encodedstore, self).datafiles():
+                    for t, a, b, size in super(encodedstore, self).datafiles():
                         try:
                             a = decodefilename(a)
                         except KeyError:
                             a = None
                         if a is not None and not _matchtrackedpath(a, matcher):
                             continue
-                        yield a, b, size
+                        yield t, a, b, size
                 def join(self, f):
                     return self.path + b'/' + encodefilename(f)
                 def copylist(self):
                     return [b'requires', b'00changelog.i'] + [b'store/' + f for f in _data]
             class fncache(object):
                 # the filename used to be partially encoded
                 # hence the encodedir/decodedir dance
                 def __init__(self, vfs):
                     self.vfs = vfs
                     self.entries = None
                     self._dirty = False
                     # set of new additions to fncache
                     self.addls = set()
                 def ensureloaded(self, warn=None):
                     """read the fncache file if not already read.
                     If the file on disk is corrupted, raise. If warn is provided,
                     warn and keep going instead."""
                     if self.entries is None:
                         self._load(warn)
                 def _load(self, warn=None):
                     '''fill the entries from the fncache file'''
                     self._dirty = False
                     try:
                         fp = self.vfs(b'fncache', mode=b'rb')
                     except IOError:
                         # skip nonexistent file
                         self.entries = set()
                         return
                     self.entries = set()
                     chunk = b''
                     for c in iter(functools.partial(fp.read, fncache_chunksize), b''):
                         chunk += c
                         try:
                             p = chunk.rindex(b'\n')
                             self.entries.update(decodedir(chunk[: p + 1]).splitlines())
                             chunk = chunk[p + 1 :]
                         except ValueError:
                             # substring '\n' not found, maybe the entry is bigger than the
                             # chunksize, so let's keep iterating
                             pass
                     if chunk:
                         msg = _(b"fncache does not ends with a newline")
                         if warn:
                             warn(msg + b'\n')
                         else:
                             raise error.Abort(
                                 msg,
                                 hint=_(
                                     b"use 'hg debugrebuildfncache' to "
                                     b"rebuild the fncache"
                                 ),
                             )
                     self._checkentries(fp, warn)
                     fp.close()
                 def _checkentries(self, fp, warn):
                     """ make sure there is no empty string in entries """
                     if b'' in self.entries:
                         fp.seek(0)
                         for n, line in enumerate(util.iterfile(fp)):
                             if not line.rstrip(b'\n'):
                                 t = _(b'invalid entry in fncache, line %d') % (n + 1)
                                 if warn:
                                     warn(t + b'\n')
                                 else:
                                     raise error.Abort(t)
                 def write(self, tr):
                     if self._dirty:
                         assert self.entries is not None
                         self.entries = self.entries | self.addls
                         self.addls = set()
                         tr.addbackup(b'fncache')
                         fp = self.vfs(b'fncache', mode=b'wb', atomictemp=True)
                         if self.entries:
                             fp.write(encodedir(b'\n'.join(self.entries) + b'\n'))
                         fp.close()
                         self._dirty = False
                     if self.addls:
                         # if we have just new entries, let's append them to the fncache
                         tr.addbackup(b'fncache')
                         fp = self.vfs(b'fncache', mode=b'ab', atomictemp=True)
                         if self.addls:
                             fp.write(encodedir(b'\n'.join(self.addls) + b'\n'))
                         fp.close()
                         self.entries = None
                         self.addls = set()
                 def add(self, fn):
                     if self.entries is None:
                         self._load()
                     if fn not in self.entries:
                         self.addls.add(fn)
                 def remove(self, fn):
                     if self.entries is None:
                         self._load()
                     if fn in self.addls:
                         self.addls.remove(fn)
                         return
                     try:
                         self.entries.remove(fn)
                         self._dirty = True
                     except KeyError:
                         pass
                 def __contains__(self, fn):
                     if fn in self.addls:
                         return True
                     if self.entries is None:
                         self._load()
                     return fn in self.entries
                 def __iter__(self):
                     if self.entries is None:
                         self._load()
                     return iter(self.entries | self.addls)
             class _fncachevfs(vfsmod.proxyvfs):
                 def __init__(self, vfs, fnc, encode):
                     vfsmod.proxyvfs.__init__(self, vfs)
                     self.fncache = fnc
                     self.encode = encode
                 def __call__(self, path, mode=b'r', *args, **kw):
                     encoded = self.encode(path)
                     if mode not in (b'r', b'rb') and (
                         path.startswith(b'data/') or path.startswith(b'meta/')
                     ):
                         # do not trigger a fncache load when adding a file that already is
                         # known to exist.
                         notload = self.fncache.entries is None and self.vfs.exists(encoded)
                         if notload and b'a' in mode and not self.vfs.stat(encoded).st_size:
                             # when appending to an existing file, if the file has size zero,
                             # it should be considered as missing. Such zero-size files are
                             # the result of truncation when a transaction is aborted.
                             notload = False
                         if not notload:
                             self.fncache.add(path)
                     return self.vfs(encoded, mode, *args, **kw)
                 def join(self, path):
                     if path:
                         return self.vfs.join(self.encode(path))
                     else:
                         return self.vfs.join(path)
             class fncachestore(basicstore):
                 def __init__(self, path, vfstype, dotencode):
                     if dotencode:
                         encode = _pathencode
                     else:
                         encode = _plainhybridencode
                     self.encode = encode
                     vfs = vfstype(path + b'/store')
                     self.path = vfs.base
                     self.pathsep = self.path + b'/'
                     self.createmode = _calcmode(vfs)
                     vfs.createmode = self.createmode
                     self.rawvfs = vfs
                     fnc = fncache(vfs)
                     self.fncache = fnc
                     self.vfs = _fncachevfs(vfs, fnc, encode)
                     self.opener = self.vfs
                 def join(self, f):
                     return self.pathsep + self.encode(f)
                 def getsize(self, path):
                     return self.rawvfs.stat(path).st_size
                 def datafiles(self, matcher=None):
                     for f in sorted(self.fncache):
                         if not _matchtrackedpath(f, matcher):
                             continue
                         ef = self.encode(f)
                         try:
-                            yield f, ef, self.getsize(ef)
+                            t = revlog_type(f)
+                            t |= FILEFLAGS_FILELOG
+                            yield t, f, ef, self.getsize(ef)
                         except OSError as err:
                             if err.errno != errno.ENOENT:
                                 raise
                 def copylist(self):
                     d = (
                         b'bookmarks',
                         b'narrowspec',
                         b'data',
                         b'meta',
                         b'dh',
                         b'fncache',
                         b'phaseroots',
                         b'obsstore',
                         b'00manifest.d',
                         b'00manifest.i',
                         b'00changelog.d',
                         b'00changelog.i',
                         b'requires',
                     )
                     return [b'requires', b'00changelog.i'] + [b'store/' + f for f in d]
                 def write(self, tr):
                     self.fncache.write(tr)
                 def invalidatecaches(self):
                     self.fncache.entries = None
                     self.fncache.addls = set()
                 def markremoved(self, fn):
                     self.fncache.remove(fn)
                 def _exists(self, f):
                     ef = self.encode(f)
                     try:
                         self.getsize(ef)
                         return True
                     except OSError as err:
                         if err.errno != errno.ENOENT:
                             raise
                         # nonexistent entry
                         return False
                 def __contains__(self, path):
                     '''Checks if the store contains path'''
                     path = b"/".join((b"data", path))
                     # check for files (exact match)
                     e = path + b'.i'
                     if e in self.fncache and self._exists(e):
                         return True
                     # now check for directories (prefix match)
                     if not path.endswith(b'/'):
                         path += b'/'
                     for e in self.fncache:
                         if e.startswith(path) and self._exists(e):
                             return True
                     return False

mercurial/streamclone.py

0 +2 -2

             # streamclone.py - producing and consuming streaming repository data
             #
             # Copyright 2015 Gregory Szorc <gregory.szorc@gmail.com>
             #
             # This software may be used and distributed according to the terms of the
             # GNU General Public License version 2 or any later version.
             from __future__ import absolute_import
             import contextlib
             import os
             import struct
             from .i18n import _
             from .pycompat import open
             from .interfaces import repository
             from . import (
                 cacheutil,
                 error,
                 narrowspec,
                 phases,
                 pycompat,
                 requirements as requirementsmod,
                 scmutil,
                 store,
                 util,
             )
             def canperformstreamclone(pullop, bundle2=False):
                 """Whether it is possible to perform a streaming clone as part of pull.
                 ``bundle2`` will cause the function to consider stream clone through
                 bundle2 and only through bundle2.
                 Returns a tuple of (supported, requirements). ``supported`` is True if
                 streaming clone is supported and False otherwise. ``requirements`` is
                 a set of repo requirements from the remote, or ``None`` if stream clone
                 isn't supported.
                 """
                 repo = pullop.repo
                 remote = pullop.remote
                 bundle2supported = False
                 if pullop.canusebundle2:
                     if b'v2' in pullop.remotebundle2caps.get(b'stream', []):
                         bundle2supported = True
                     # else
                     # Server doesn't support bundle2 stream clone or doesn't support
                     # the versions we support. Fall back and possibly allow legacy.
                 # Ensures legacy code path uses available bundle2.
                 if bundle2supported and not bundle2:
                     return False, None
                 # Ensures bundle2 doesn't try to do a stream clone if it isn't supported.
                 elif bundle2 and not bundle2supported:
                     return False, None
                 # Streaming clone only works on empty repositories.
                 if len(repo):
                     return False, None
                 # Streaming clone only works if all data is being requested.
                 if pullop.heads:
                     return False, None
                 streamrequested = pullop.streamclonerequested
                 # If we don't have a preference, let the server decide for us. This
                 # likely only comes into play in LANs.
                 if streamrequested is None:
                     # The server can advertise whether to prefer streaming clone.
                     streamrequested = remote.capable(b'stream-preferred')
                 if not streamrequested:
                     return False, None
                 # In order for stream clone to work, the client has to support all the
                 # requirements advertised by the server.
                 #
                 # The server advertises its requirements via the "stream" and "streamreqs"
                 # capability. "stream" (a value-less capability) is advertised if and only
                 # if the only requirement is "revlogv1." Else, the "streamreqs" capability
                 # is advertised and contains a comma-delimited list of requirements.
                 requirements = set()
                 if remote.capable(b'stream'):
                     requirements.add(requirementsmod.REVLOGV1_REQUIREMENT)
                 else:
                     streamreqs = remote.capable(b'streamreqs')
                     # This is weird and shouldn't happen with modern servers.
                     if not streamreqs:
                         pullop.repo.ui.warn(
                             _(
                                 b'warning: stream clone requested but server has them '
                                 b'disabled\n'
                             )
                         )
                         return False, None
                     streamreqs = set(streamreqs.split(b','))
                     # Server requires something we don't support. Bail.
                     missingreqs = streamreqs - repo.supportedformats
                     if missingreqs:
                         pullop.repo.ui.warn(
                             _(
                                 b'warning: stream clone requested but client is missing '
                                 b'requirements: %s\n'
                             )
                             % b', '.join(sorted(missingreqs))
                         )
                         pullop.repo.ui.warn(
                             _(
                                 b'(see https://www.mercurial-scm.org/wiki/MissingRequirement '
                                 b'for more information)\n'
                             )
                         )
                         return False, None
                     requirements = streamreqs
                 return True, requirements
             def maybeperformlegacystreamclone(pullop):
                 """Possibly perform a legacy stream clone operation.
                 Legacy stream clones are performed as part of pull but before all other
                 operations.
                 A legacy stream clone will not be performed if a bundle2 stream clone is
                 supported.
                 """
                 from . import localrepo
                 supported, requirements = canperformstreamclone(pullop)
                 if not supported:
                     return
                 repo = pullop.repo
                 remote = pullop.remote
                 # Save remote branchmap. We will use it later to speed up branchcache
                 # creation.
                 rbranchmap = None
                 if remote.capable(b'branchmap'):
                     with remote.commandexecutor() as e:
                         rbranchmap = e.callcommand(b'branchmap', {}).result()
                 repo.ui.status(_(b'streaming all changes\n'))
                 with remote.commandexecutor() as e:
                     fp = e.callcommand(b'stream_out', {}).result()
                 # TODO strictly speaking, this code should all be inside the context
                 # manager because the context manager is supposed to ensure all wire state
                 # is flushed when exiting. But the legacy peers don't do this, so it
                 # doesn't matter.
                 l = fp.readline()
                 try:
                     resp = int(l)
                 except ValueError:
                     raise error.ResponseError(
                         _(b'unexpected response from remote server:'), l
                     )
                 if resp == 1:
                     raise error.Abort(_(b'operation forbidden by server'))
                 elif resp == 2:
                     raise error.Abort(_(b'locking the remote repository failed'))
                 elif resp != 0:
                     raise error.Abort(_(b'the server sent an unknown error code'))
                 l = fp.readline()
                 try:
                     filecount, bytecount = map(int, l.split(b' ', 1))
                 except (ValueError, TypeError):
                     raise error.ResponseError(
                         _(b'unexpected response from remote server:'), l
                     )
                 with repo.lock():
                     consumev1(repo, fp, filecount, bytecount)
                     # new requirements = old non-format requirements +
                     #                    new format-related remote requirements
                     # requirements from the streamed-in repository
                     repo.requirements = requirements | (
                         repo.requirements - repo.supportedformats
                     )
                     repo.svfs.options = localrepo.resolvestorevfsoptions(
                         repo.ui, repo.requirements, repo.features
                     )
                     scmutil.writereporequirements(repo)
                     if rbranchmap:
                         repo._branchcaches.replace(repo, rbranchmap)
                     repo.invalidate()
             def allowservergeneration(repo):
                 """Whether streaming clones are allowed from the server."""
                 if repository.REPO_FEATURE_STREAM_CLONE not in repo.features:
                     return False
                 if not repo.ui.configbool(b'server', b'uncompressed', untrusted=True):
                     return False
                 # The way stream clone works makes it impossible to hide secret changesets.
                 # So don't allow this by default.
                 secret = phases.hassecret(repo)
                 if secret:
                     return repo.ui.configbool(b'server', b'uncompressedallowsecret')
                 return True
             # This is it's own function so extensions can override it.
             def _walkstreamfiles(repo, matcher=None):
                 return repo.store.walk(matcher)
             def generatev1(repo):
                 """Emit content for version 1 of a streaming clone.
                 This returns a 3-tuple of (file count, byte size, data iterator).
                 The data iterator consists of N entries for each file being transferred.
                 Each file entry starts as a line with the file name and integer size
                 delimited by a null byte.
                 The raw file data follows. Following the raw file data is the next file
                 entry, or EOF.
                 When used on the wire protocol, an additional line indicating protocol
                 success will be prepended to the stream. This function is not responsible
                 for adding it.
                 This function will obtain a repository lock to ensure a consistent view of
                 the store is captured. It therefore may raise LockError.
                 """
                 entries = []
                 total_bytes = 0
                 # Get consistent snapshot of repo, lock during scan.
                 with repo.lock():
                     repo.ui.debug(b'scanning\n')
-                    for name, ename, size in _walkstreamfiles(repo):
+                    for file_type, name, ename, size in _walkstreamfiles(repo):
                         if size:
                             entries.append((name, size))
                             total_bytes += size
                 repo.ui.debug(
                     b'%d files, %d bytes to transfer\n' % (len(entries), total_bytes)
                 )
                 svfs = repo.svfs
                 debugflag = repo.ui.debugflag
                 def emitrevlogdata():
                     for name, size in entries:
                         if debugflag:
                             repo.ui.debug(b'sending %s (%d bytes)\n' % (name, size))
                         # partially encode name over the wire for backwards compat
                         yield b'%s\0%d\n' % (store.encodedir(name), size)
                         # auditing at this stage is both pointless (paths are already
                         # trusted by the local repo) and expensive
                         with svfs(name, b'rb', auditpath=False) as fp:
                             if size <= 65536:
                                 yield fp.read(size)
                             else:
                                 for chunk in util.filechunkiter(fp, limit=size):
                                     yield chunk
                 return len(entries), total_bytes, emitrevlogdata()
             def generatev1wireproto(repo):
                 """Emit content for version 1 of streaming clone suitable for the wire.
                 This is the data output from ``generatev1()`` with 2 header lines. The
                 first line indicates overall success. The 2nd contains the file count and
                 byte size of payload.
                 The success line contains "0" for success, "1" for stream generation not
                 allowed, and "2" for error locking the repository (possibly indicating
                 a permissions error for the server process).
                 """
                 if not allowservergeneration(repo):
                     yield b'1\n'
                     return
                 try:
                     filecount, bytecount, it = generatev1(repo)
                 except error.LockError:
                     yield b'2\n'
                     return
                 # Indicates successful response.
                 yield b'0\n'
                 yield b'%d %d\n' % (filecount, bytecount)
                 for chunk in it:
                     yield chunk
             def generatebundlev1(repo, compression=b'UN'):
                 """Emit content for version 1 of a stream clone bundle.
                 The first 4 bytes of the output ("HGS1") denote this as stream clone
                 bundle version 1.
                 The next 2 bytes indicate the compression type. Only "UN" is currently
                 supported.
                 The next 16 bytes are two 64-bit big endian unsigned integers indicating
                 file count and byte count, respectively.
                 The next 2 bytes is a 16-bit big endian unsigned short declaring the length
                 of the requirements string, including a trailing \0. The following N bytes
                 are the requirements string, which is ASCII containing a comma-delimited
                 list of repo requirements that are needed to support the data.
                 The remaining content is the output of ``generatev1()`` (which may be
                 compressed in the future).
                 Returns a tuple of (requirements, data generator).
                 """
                 if compression != b'UN':
                     raise ValueError(b'we do not support the compression argument yet')
                 requirements = repo.requirements & repo.supportedformats
                 requires = b','.join(sorted(requirements))
                 def gen():
                     yield b'HGS1'
                     yield compression
                     filecount, bytecount, it = generatev1(repo)
                     repo.ui.status(
                         _(b'writing %d bytes for %d files\n') % (bytecount, filecount)
                     )
                     yield struct.pack(b'>QQ', filecount, bytecount)
                     yield struct.pack(b'>H', len(requires) + 1)
                     yield requires + b'\0'
                     # This is where we'll add compression in the future.
                     assert compression == b'UN'
                     progress = repo.ui.makeprogress(
                         _(b'bundle'), total=bytecount, unit=_(b'bytes')
                     )
                     progress.update(0)
                     for chunk in it:
                         progress.increment(step=len(chunk))
                         yield chunk
                     progress.complete()
                 return requirements, gen()
             def consumev1(repo, fp, filecount, bytecount):
                 """Apply the contents from version 1 of a streaming clone file handle.
                 This takes the output from "stream_out" and applies it to the specified
                 repository.
                 Like "stream_out," the status line added by the wire protocol is not
                 handled by this function.
                 """
                 with repo.lock():
                     repo.ui.status(
                         _(b'%d files to transfer, %s of data\n')
                         % (filecount, util.bytecount(bytecount))
                     )
                     progress = repo.ui.makeprogress(
                         _(b'clone'), total=bytecount, unit=_(b'bytes')
                     )
                     progress.update(0)
                     start = util.timer()
                     # TODO: get rid of (potential) inconsistency
                     #
                     # If transaction is started and any @filecache property is
                     # changed at this point, it causes inconsistency between
                     # in-memory cached property and streamclone-ed file on the
                     # disk. Nested transaction prevents transaction scope "clone"
                     # below from writing in-memory changes out at the end of it,
                     # even though in-memory changes are discarded at the end of it
                     # regardless of transaction nesting.
                     #
                     # But transaction nesting can't be simply prohibited, because
                     # nesting occurs also in ordinary case (e.g. enabling
                     # clonebundles).
                     with repo.transaction(b'clone'):
                         with repo.svfs.backgroundclosing(repo.ui, expectedcount=filecount):
                             for i in pycompat.xrange(filecount):
                                 # XXX doesn't support '\n' or '\r' in filenames
                                 l = fp.readline()
                                 try:
                                     name, size = l.split(b'\0', 1)
                                     size = int(size)
                                 except (ValueError, TypeError):
                                     raise error.ResponseError(
                                         _(b'unexpected response from remote server:'), l
                                     )
                                 if repo.ui.debugflag:
                                     repo.ui.debug(
                                         b'adding %s (%s)\n' % (name, util.bytecount(size))
                                     )
                                 # for backwards compat, name was partially encoded
                                 path = store.decodedir(name)
                                 with repo.svfs(path, b'w', backgroundclose=True) as ofp:
                                     for chunk in util.filechunkiter(fp, limit=size):
                                         progress.increment(step=len(chunk))
                                         ofp.write(chunk)
                         # force @filecache properties to be reloaded from
                         # streamclone-ed file at next access
                         repo.invalidate(clearfilecache=True)
                     elapsed = util.timer() - start
                     if elapsed <= 0:
                         elapsed = 0.001
                     progress.complete()
                     repo.ui.status(
                         _(b'transferred %s in %.1f seconds (%s/sec)\n')
                         % (
                             util.bytecount(bytecount),
                             elapsed,
                             util.bytecount(bytecount / elapsed),
                         )
                     )
             def readbundle1header(fp):
                 compression = fp.read(2)
                 if compression != b'UN':
                     raise error.Abort(
                         _(
                             b'only uncompressed stream clone bundles are '
                             b'supported; got %s'
                         )
                         % compression
                     )
                 filecount, bytecount = struct.unpack(b'>QQ', fp.read(16))
                 requireslen = struct.unpack(b'>H', fp.read(2))[0]
                 requires = fp.read(requireslen)
                 if not requires.endswith(b'\0'):
                     raise error.Abort(
                         _(
                             b'malformed stream clone bundle: '
                             b'requirements not properly encoded'
                         )
                     )
                 requirements = set(requires.rstrip(b'\0').split(b','))
                 return filecount, bytecount, requirements
             def applybundlev1(repo, fp):
                 """Apply the content from a stream clone bundle version 1.
                 We assume the 4 byte header has been read and validated and the file handle
                 is at the 2 byte compression identifier.
                 """
                 if len(repo):
                     raise error.Abort(
                         _(b'cannot apply stream clone bundle on non-empty repo')
                     )
                 filecount, bytecount, requirements = readbundle1header(fp)
                 missingreqs = requirements - repo.supportedformats
                 if missingreqs:
                     raise error.Abort(
                         _(b'unable to apply stream clone: unsupported format: %s')
                         % b', '.join(sorted(missingreqs))
                     )
                 consumev1(repo, fp, filecount, bytecount)
             class streamcloneapplier(object):
                 """Class to manage applying streaming clone bundles.
                 We need to wrap ``applybundlev1()`` in a dedicated type to enable bundle
                 readers to perform bundle type-specific functionality.
                 """
                 def __init__(self, fh):
                     self._fh = fh
                 def apply(self, repo):
                     return applybundlev1(repo, self._fh)
             # type of file to stream
             _fileappend = 0  # append only file
             _filefull = 1  # full snapshot file
             # Source of the file
             _srcstore = b's'  # store (svfs)
             _srccache = b'c'  # cache (cache)
             # This is it's own function so extensions can override it.
             def _walkstreamfullstorefiles(repo):
                 """list snapshot file from the store"""
                 fnames = []
                 if not repo.publishing():
                     fnames.append(b'phaseroots')
                 return fnames
             def _filterfull(entry, copy, vfsmap):
                 """actually copy the snapshot files"""
                 src, name, ftype, data = entry
                 if ftype != _filefull:
                     return entry
                 return (src, name, ftype, copy(vfsmap[src].join(name)))
             @contextlib.contextmanager
             def maketempcopies():
                 """return a function to temporary copy file"""
                 files = []
                 try:
                     def copy(src):
                         fd, dst = pycompat.mkstemp()
                         os.close(fd)
                         files.append(dst)
                         util.copyfiles(src, dst, hardlink=True)
                         return dst
                     yield copy
                 finally:
                     for tmp in files:
                         util.tryunlink(tmp)
             def _makemap(repo):
                 """make a (src -> vfs) map for the repo"""
                 vfsmap = {
                     _srcstore: repo.svfs,
                     _srccache: repo.cachevfs,
                 }
                 # we keep repo.vfs out of the on purpose, ther are too many danger there
                 # (eg: .hg/hgrc)
                 assert repo.vfs not in vfsmap.values()
                 return vfsmap
             def _emit2(repo, entries, totalfilesize):
                 """actually emit the stream bundle"""
                 vfsmap = _makemap(repo)
                 progress = repo.ui.makeprogress(
                     _(b'bundle'), total=totalfilesize, unit=_(b'bytes')
                 )
                 progress.update(0)
                 with maketempcopies() as copy, progress:
                     # copy is delayed until we are in the try
                     entries = [_filterfull(e, copy, vfsmap) for e in entries]
                     yield None  # this release the lock on the repository
                     seen = 0
                     for src, name, ftype, data in entries:
                         vfs = vfsmap[src]
                         yield src
                         yield util.uvarintencode(len(name))
                         if ftype == _fileappend:
                             fp = vfs(name)
                             size = data
                         elif ftype == _filefull:
                             fp = open(data, b'rb')
                             size = util.fstat(fp).st_size
                         try:
                             yield util.uvarintencode(size)
                             yield name
                             if size <= 65536:
                                 chunks = (fp.read(size),)
                             else:
                                 chunks = util.filechunkiter(fp, limit=size)
                             for chunk in chunks:
                                 seen += len(chunk)
                                 progress.update(seen)
                                 yield chunk
                         finally:
                             fp.close()
             def generatev2(repo, includes, excludes, includeobsmarkers):
                 """Emit content for version 2 of a streaming clone.
                 the data stream consists the following entries:
 ) A char representing the file destination (eg: store or cache)
 ) A varint containing the length of the filename
 ) A varint containing the length of file data
 ) N bytes containing the filename (the internal, store-agnostic form)
 ) N bytes containing the file data
                 Returns a 3-tuple of (file count, file size, data iterator).
                 """
                 with repo.lock():
                     entries = []
                     totalfilesize = 0
                     matcher = None
                     if includes or excludes:
                         matcher = narrowspec.match(repo.root, includes, excludes)
                     repo.ui.debug(b'scanning\n')
-                    for name, ename, size in _walkstreamfiles(repo, matcher):
+                    for rl_type, name, ename, size in _walkstreamfiles(repo, matcher):
                         if size:
                             entries.append((_srcstore, name, _fileappend, size))
                             totalfilesize += size
                     for name in _walkstreamfullstorefiles(repo):
                         if repo.svfs.exists(name):
                             totalfilesize += repo.svfs.lstat(name).st_size
                             entries.append((_srcstore, name, _filefull, None))
                     if includeobsmarkers and repo.svfs.exists(b'obsstore'):
                         totalfilesize += repo.svfs.lstat(b'obsstore').st_size
                         entries.append((_srcstore, b'obsstore', _filefull, None))
                     for name in cacheutil.cachetocopy(repo):
                         if repo.cachevfs.exists(name):
                             totalfilesize += repo.cachevfs.lstat(name).st_size
                             entries.append((_srccache, name, _filefull, None))
                     chunks = _emit2(repo, entries, totalfilesize)
                     first = next(chunks)
                     assert first is None
                 return len(entries), totalfilesize, chunks
             @contextlib.contextmanager
             def nested(*ctxs):
                 this = ctxs[0]
                 rest = ctxs[1:]
                 with this:
                     if rest:
                         with nested(*rest):
                             yield
                     else:
                         yield
             def consumev2(repo, fp, filecount, filesize):
                 """Apply the contents from a version 2 streaming clone.
                 Data is read from an object that only needs to provide a ``read(size)``
                 method.
                 """
                 with repo.lock():
                     repo.ui.status(
                         _(b'%d files to transfer, %s of data\n')
                         % (filecount, util.bytecount(filesize))
                     )
                     start = util.timer()
                     progress = repo.ui.makeprogress(
                         _(b'clone'), total=filesize, unit=_(b'bytes')
                     )
                     progress.update(0)
                     vfsmap = _makemap(repo)
                     with repo.transaction(b'clone'):
                         ctxs = (vfs.backgroundclosing(repo.ui) for vfs in vfsmap.values())
                         with nested(*ctxs):
                             for i in range(filecount):
                                 src = util.readexactly(fp, 1)
                                 vfs = vfsmap[src]
                                 namelen = util.uvarintdecodestream(fp)
                                 datalen = util.uvarintdecodestream(fp)
                                 name = util.readexactly(fp, namelen)
                                 if repo.ui.debugflag:
                                     repo.ui.debug(
                                         b'adding [%s] %s (%s)\n'
                                         % (src, name, util.bytecount(datalen))
                                     )
                                 with vfs(name, b'w') as ofp:
                                     for chunk in util.filechunkiter(fp, limit=datalen):
                                         progress.increment(step=len(chunk))
                                         ofp.write(chunk)
                         # force @filecache properties to be reloaded from
                         # streamclone-ed file at next access
                         repo.invalidate(clearfilecache=True)
                     elapsed = util.timer() - start
                     if elapsed <= 0:
                         elapsed = 0.001
                     repo.ui.status(
                         _(b'transferred %s in %.1f seconds (%s/sec)\n')
                         % (
                             util.bytecount(progress.pos),
                             elapsed,
                             util.bytecount(progress.pos / elapsed),
                         )
                     )
                     progress.complete()
             def applybundlev2(repo, fp, filecount, filesize, requirements):
                 from . import localrepo
                 missingreqs = [r for r in requirements if r not in repo.supported]
                 if missingreqs:
                     raise error.Abort(
                         _(b'unable to apply stream clone: unsupported format: %s')
                         % b', '.join(sorted(missingreqs))
                     )
                 consumev2(repo, fp, filecount, filesize)
                 # new requirements = old non-format requirements +
                 #                    new format-related remote requirements
                 # requirements from the streamed-in repository
                 repo.requirements = set(requirements) | (
                     repo.requirements - repo.supportedformats
                 )
                 repo.svfs.options = localrepo.resolvestorevfsoptions(
                     repo.ui, repo.requirements, repo.features
                 )
                 scmutil.writereporequirements(repo)

mercurial/upgrade_utils/engine.py

0 +1 -1

             # upgrade.py - functions for in place upgrade of Mercurial repository
             #
             # Copyright (c) 2016-present, Gregory Szorc
             #
             # This software may be used and distributed according to the terms of the
             # GNU General Public License version 2 or any later version.
             from __future__ import absolute_import
             import stat
             from ..i18n import _
             from ..pycompat import getattr
             from .. import (
                 changelog,
                 error,
                 filelog,
                 manifest,
                 metadata,
                 pycompat,
                 requirements,
                 revlog,
                 scmutil,
                 util,
                 vfs as vfsmod,
             )
             from ..revlogutils import nodemap
             def _revlogfrompath(repo, path):
                 """Obtain a revlog from a repo path.
                 An instance of the appropriate class is returned.
                 """
                 if path == b'00changelog.i':
                     return changelog.changelog(repo.svfs)
                 elif path.endswith(b'00manifest.i'):
                     mandir = path[: -len(b'00manifest.i')]
                     return manifest.manifestrevlog(
                         repo.nodeconstants, repo.svfs, tree=mandir
                     )
                 else:
                     # reverse of "/".join(("data", path + ".i"))
                     return filelog.filelog(repo.svfs, path[5:-2])
             def _copyrevlog(tr, destrepo, oldrl, unencodedname):
                 """copy all relevant files for `oldrl` into `destrepo` store
                 Files are copied "as is" without any transformation. The copy is performed
                 without extra checks. Callers are responsible for making sure the copied
                 content is compatible with format of the destination repository.
                 """
                 oldrl = getattr(oldrl, '_revlog', oldrl)
                 newrl = _revlogfrompath(destrepo, unencodedname)
                 newrl = getattr(newrl, '_revlog', newrl)
                 oldvfs = oldrl.opener
                 newvfs = newrl.opener
                 oldindex = oldvfs.join(oldrl.indexfile)
                 newindex = newvfs.join(newrl.indexfile)
                 olddata = oldvfs.join(oldrl.datafile)
                 newdata = newvfs.join(newrl.datafile)
                 with newvfs(newrl.indexfile, b'w'):
                     pass  # create all the directories
                 util.copyfile(oldindex, newindex)
                 copydata = oldrl.opener.exists(oldrl.datafile)
                 if copydata:
                     util.copyfile(olddata, newdata)
                 if not (
                     unencodedname.endswith(b'00changelog.i')
                     or unencodedname.endswith(b'00manifest.i')
                 ):
                     destrepo.svfs.fncache.add(unencodedname)
                     if copydata:
                         destrepo.svfs.fncache.add(unencodedname[:-2] + b'.d')
             UPGRADE_CHANGELOG = b"changelog"
             UPGRADE_MANIFEST = b"manifest"
             UPGRADE_FILELOGS = b"all-filelogs"
             UPGRADE_ALL_REVLOGS = frozenset(
                 [UPGRADE_CHANGELOG, UPGRADE_MANIFEST, UPGRADE_FILELOGS]
             )
             def getsidedatacompanion(srcrepo, dstrepo):
                 sidedatacompanion = None
                 removedreqs = srcrepo.requirements - dstrepo.requirements
                 addedreqs = dstrepo.requirements - srcrepo.requirements
                 if requirements.SIDEDATA_REQUIREMENT in removedreqs:
                     def sidedatacompanion(rl, rev):
                         rl = getattr(rl, '_revlog', rl)
                         if rl.flags(rev) & revlog.REVIDX_SIDEDATA:
                             return True, (), {}, 0, 0
                         return False, (), {}, 0, 0
                 elif requirements.COPIESSDC_REQUIREMENT in addedreqs:
                     sidedatacompanion = metadata.getsidedataadder(srcrepo, dstrepo)
                 elif requirements.COPIESSDC_REQUIREMENT in removedreqs:
                     sidedatacompanion = metadata.getsidedataremover(srcrepo, dstrepo)
                 return sidedatacompanion
             def matchrevlog(revlogfilter, entry):
                 """check if a revlog is selected for cloning.
                 In other words, are there any updates which need to be done on revlog
                 or it can be blindly copied.
                 The store entry is checked against the passed filter"""
                 if entry.endswith(b'00changelog.i'):
                     return UPGRADE_CHANGELOG in revlogfilter
                 elif entry.endswith(b'00manifest.i'):
                     return UPGRADE_MANIFEST in revlogfilter
                 return UPGRADE_FILELOGS in revlogfilter
             def _perform_clone(
                 ui,
                 dstrepo,
                 tr,
                 old_revlog,
                 unencoded,
                 upgrade_op,
                 sidedatacompanion,
                 oncopiedrevision,
             ):
                 """ returns the new revlog object created"""
                 newrl = None
                 if matchrevlog(upgrade_op.revlogs_to_process, unencoded):
                     ui.note(
                         _(b'cloning %d revisions from %s\n') % (len(old_revlog), unencoded)
                     )
                     newrl = _revlogfrompath(dstrepo, unencoded)
                     old_revlog.clone(
                         tr,
                         newrl,
                         addrevisioncb=oncopiedrevision,
                         deltareuse=upgrade_op.delta_reuse_mode,
                         forcedeltabothparents=upgrade_op.force_re_delta_both_parents,
                         sidedatacompanion=sidedatacompanion,
                     )
                 else:
                     msg = _(b'blindly copying %s containing %i revisions\n')
                     ui.note(msg % (unencoded, len(old_revlog)))
                     _copyrevlog(tr, dstrepo, old_revlog, unencoded)
                     newrl = _revlogfrompath(dstrepo, unencoded)
                 return newrl
             def _clonerevlogs(
                 ui,
                 srcrepo,
                 dstrepo,
                 tr,
                 upgrade_op,
             ):
                 """Copy revlogs between 2 repos."""
                 revcount = 0
                 srcsize = 0
                 srcrawsize = 0
                 dstsize = 0
                 fcount = 0
                 frevcount = 0
                 fsrcsize = 0
                 frawsize = 0
                 fdstsize = 0
                 mcount = 0
                 mrevcount = 0
                 msrcsize = 0
                 mrawsize = 0
                 mdstsize = 0
                 crevcount = 0
                 csrcsize = 0
                 crawsize = 0
                 cdstsize = 0
                 alldatafiles = list(srcrepo.store.walk())
                 # mapping of data files which needs to be cloned
                 # key is unencoded filename
                 # value is revlog_object_from_srcrepo
                 manifests = {}
                 changelogs = {}
                 filelogs = {}
                 # Perform a pass to collect metadata. This validates we can open all
                 # source files and allows a unified progress bar to be displayed.
-                for unencoded, encoded, size in alldatafiles:
+                for revlog_type, unencoded, encoded, size in alldatafiles:
                     if not unencoded.endswith(b'.i'):
                         continue
                     rl = _revlogfrompath(srcrepo, unencoded)
                     info = rl.storageinfo(
                         exclusivefiles=True,
                         revisionscount=True,
                         trackedsize=True,
                         storedsize=True,
                     )
                     revcount += info[b'revisionscount'] or 0
                     datasize = info[b'storedsize'] or 0
                     rawsize = info[b'trackedsize'] or 0
                     srcsize += datasize
                     srcrawsize += rawsize
                     # This is for the separate progress bars.
                     if isinstance(rl, changelog.changelog):
                         changelogs[unencoded] = rl
                         crevcount += len(rl)
                         csrcsize += datasize
                         crawsize += rawsize
                     elif isinstance(rl, manifest.manifestrevlog):
                         manifests[unencoded] = rl
                         mcount += 1
                         mrevcount += len(rl)
                         msrcsize += datasize
                         mrawsize += rawsize
                     elif isinstance(rl, filelog.filelog):
                         filelogs[unencoded] = rl
                         fcount += 1
                         frevcount += len(rl)
                         fsrcsize += datasize
                         frawsize += rawsize
                     else:
                         error.ProgrammingError(b'unknown revlog type')
                 if not revcount:
                     return
                 ui.status(
                     _(
                         b'migrating %d total revisions (%d in filelogs, %d in manifests, '
                         b'%d in changelog)\n'
                     )
                     % (revcount, frevcount, mrevcount, crevcount)
                 )
                 ui.status(
                     _(b'migrating %s in store; %s tracked data\n')
                     % ((util.bytecount(srcsize), util.bytecount(srcrawsize)))
                 )
                 # Used to keep track of progress.
                 progress = None
                 def oncopiedrevision(rl, rev, node):
                     progress.increment()
                 sidedatacompanion = getsidedatacompanion(srcrepo, dstrepo)
                 # Migrating filelogs
                 ui.status(
                     _(
                         b'migrating %d filelogs containing %d revisions '
                         b'(%s in store; %s tracked data)\n'
                     )
                     % (
                         fcount,
                         frevcount,
                         util.bytecount(fsrcsize),
                         util.bytecount(frawsize),
                     )
                 )
                 progress = srcrepo.ui.makeprogress(_(b'file revisions'), total=frevcount)
                 for unencoded, oldrl in sorted(filelogs.items()):
                     newrl = _perform_clone(
                         ui,
                         dstrepo,
                         tr,
                         oldrl,
                         unencoded,
                         upgrade_op,
                         sidedatacompanion,
                         oncopiedrevision,
                     )
                     info = newrl.storageinfo(storedsize=True)
                     fdstsize += info[b'storedsize'] or 0
                 ui.status(
                     _(
                         b'finished migrating %d filelog revisions across %d '
                         b'filelogs; change in size: %s\n'
                     )
                     % (frevcount, fcount, util.bytecount(fdstsize - fsrcsize))
                 )
                 # Migrating manifests
                 ui.status(
                     _(
                         b'migrating %d manifests containing %d revisions '
                         b'(%s in store; %s tracked data)\n'
                     )
                     % (
                         mcount,
                         mrevcount,
                         util.bytecount(msrcsize),
                         util.bytecount(mrawsize),
                     )
                 )
                 if progress:
                     progress.complete()
                 progress = srcrepo.ui.makeprogress(
                     _(b'manifest revisions'), total=mrevcount
                 )
                 for unencoded, oldrl in sorted(manifests.items()):
                     newrl = _perform_clone(
                         ui,
                         dstrepo,
                         tr,
                         oldrl,
                         unencoded,
                         upgrade_op,
                         sidedatacompanion,
                         oncopiedrevision,
                     )
                     info = newrl.storageinfo(storedsize=True)
                     mdstsize += info[b'storedsize'] or 0
                 ui.status(
                     _(
                         b'finished migrating %d manifest revisions across %d '
                         b'manifests; change in size: %s\n'
                     )
                     % (mrevcount, mcount, util.bytecount(mdstsize - msrcsize))
                 )
                 # Migrating changelog
                 ui.status(
                     _(
                         b'migrating changelog containing %d revisions '
                         b'(%s in store; %s tracked data)\n'
                     )
                     % (
                         crevcount,
                         util.bytecount(csrcsize),
                         util.bytecount(crawsize),
                     )
                 )
                 if progress:
                     progress.complete()
                 progress = srcrepo.ui.makeprogress(
                     _(b'changelog revisions'), total=crevcount
                 )
                 for unencoded, oldrl in sorted(changelogs.items()):
                     newrl = _perform_clone(
                         ui,
                         dstrepo,
                         tr,
                         oldrl,
                         unencoded,
                         upgrade_op,
                         sidedatacompanion,
                         oncopiedrevision,
                     )
                     info = newrl.storageinfo(storedsize=True)
                     cdstsize += info[b'storedsize'] or 0
                 progress.complete()
                 ui.status(
                     _(
                         b'finished migrating %d changelog revisions; change in size: '
                         b'%s\n'
                     )
                     % (crevcount, util.bytecount(cdstsize - csrcsize))
                 )
                 dstsize = fdstsize + mdstsize + cdstsize
                 ui.status(
                     _(
                         b'finished migrating %d total revisions; total change in store '
                         b'size: %s\n'
                     )
                     % (revcount, util.bytecount(dstsize - srcsize))
                 )
             def _files_to_copy_post_revlog_clone(srcrepo):
                 """yields files which should be copied to destination after revlogs
                 are cloned"""
                 for path, kind, st in sorted(srcrepo.store.vfs.readdir(b'', stat=True)):
                     # don't copy revlogs as they are already cloned
                     if path.endswith((b'.i', b'.d', b'.n', b'.nd')):
                         continue
                     # Skip transaction related files.
                     if path.startswith(b'undo'):
                         continue
                     # Only copy regular files.
                     if kind != stat.S_IFREG:
                         continue
                     # Skip other skipped files.
                     if path in (b'lock', b'fncache'):
                         continue
                     # TODO: should we skip cache too?
                     yield path
             def _replacestores(currentrepo, upgradedrepo, backupvfs, upgrade_op):
                 """Replace the stores after current repository is upgraded
                 Creates a backup of current repository store at backup path
                 Replaces upgraded store files in current repo from upgraded one
                 Arguments:
                   currentrepo: repo object of current repository
                   upgradedrepo: repo object of the upgraded data
                   backupvfs: vfs object for the backup path
                   upgrade_op: upgrade operation object
                               to be used to decide what all is upgraded
                 """
                 # TODO: don't blindly rename everything in store
                 # There can be upgrades where store is not touched at all
                 if upgrade_op.backup_store:
                     util.rename(currentrepo.spath, backupvfs.join(b'store'))
                 else:
                     currentrepo.vfs.rmtree(b'store', forcibly=True)
                 util.rename(upgradedrepo.spath, currentrepo.spath)
             def finishdatamigration(ui, srcrepo, dstrepo, requirements):
                 """Hook point for extensions to perform additional actions during upgrade.
                 This function is called after revlogs and store files have been copied but
                 before the new store is swapped into the original location.
                 """
             def upgrade(ui, srcrepo, dstrepo, upgrade_op):
                 """Do the low-level work of upgrading a repository.
                 The upgrade is effectively performed as a copy between a source
                 repository and a temporary destination repository.
                 The source repository is unmodified for as long as possible so the
                 upgrade can abort at any time without causing loss of service for
                 readers and without corrupting the source repository.
                 """
                 assert srcrepo.currentwlock()
                 assert dstrepo.currentwlock()
                 backuppath = None
                 backupvfs = None
                 ui.status(
                     _(
                         b'(it is safe to interrupt this process any time before '
                         b'data migration completes)\n'
                     )
                 )
                 if upgrade_op.requirements_only:
                     ui.status(_(b'upgrading repository requirements\n'))
                     scmutil.writereporequirements(srcrepo, upgrade_op.new_requirements)
                 # if there is only one action and that is persistent nodemap upgrade
                 # directly write the nodemap file and update requirements instead of going
                 # through the whole cloning process
                 elif (
                     len(upgrade_op.upgrade_actions) == 1
                     and b'persistent-nodemap' in upgrade_op._upgrade_actions_names
                     and not upgrade_op.removed_actions
                 ):
                     ui.status(
                         _(b'upgrading repository to use persistent nodemap feature\n')
                     )
                     with srcrepo.transaction(b'upgrade') as tr:
                         unfi = srcrepo.unfiltered()
                         cl = unfi.changelog
                         nodemap.persist_nodemap(tr, cl, force=True)
                         # we want to directly operate on the underlying revlog to force
                         # create a nodemap file. This is fine since this is upgrade code
                         # and it heavily relies on repository being revlog based
                         # hence accessing private attributes can be justified
                         nodemap.persist_nodemap(
                             tr, unfi.manifestlog._rootstore._revlog, force=True
                         )
                     scmutil.writereporequirements(srcrepo, upgrade_op.new_requirements)
                 elif (
                     len(upgrade_op.removed_actions) == 1
                     and [
                         x
                         for x in upgrade_op.removed_actions
                         if x.name == b'persistent-nodemap'
                     ]
                     and not upgrade_op.upgrade_actions
                 ):
                     ui.status(
                         _(b'downgrading repository to not use persistent nodemap feature\n')
                     )
                     with srcrepo.transaction(b'upgrade') as tr:
                         unfi = srcrepo.unfiltered()
                         cl = unfi.changelog
                         nodemap.delete_nodemap(tr, srcrepo, cl)
                         # check comment 20 lines above for accessing private attributes
                         nodemap.delete_nodemap(
                             tr, srcrepo, unfi.manifestlog._rootstore._revlog
                         )
                     scmutil.writereporequirements(srcrepo, upgrade_op.new_requirements)
                 else:
                     with dstrepo.transaction(b'upgrade') as tr:
                         _clonerevlogs(
                             ui,
                             srcrepo,
                             dstrepo,
                             tr,
                             upgrade_op,
                         )
                     # Now copy other files in the store directory.
                     for p in _files_to_copy_post_revlog_clone(srcrepo):
                         srcrepo.ui.status(_(b'copying %s\n') % p)
                         src = srcrepo.store.rawvfs.join(p)
                         dst = dstrepo.store.rawvfs.join(p)
                         util.copyfile(src, dst, copystat=True)
                     finishdatamigration(ui, srcrepo, dstrepo, requirements)
                     ui.status(_(b'data fully upgraded in a temporary repository\n'))
                     if upgrade_op.backup_store:
                         backuppath = pycompat.mkdtemp(
                             prefix=b'upgradebackup.', dir=srcrepo.path
                         )
                         backupvfs = vfsmod.vfs(backuppath)
                         # Make a backup of requires file first, as it is the first to be modified.
                         util.copyfile(
                             srcrepo.vfs.join(b'requires'), backupvfs.join(b'requires')
                         )
                     # We install an arbitrary requirement that clients must not support
                     # as a mechanism to lock out new clients during the data swap. This is
                     # better than allowing a client to continue while the repository is in
                     # an inconsistent state.
                     ui.status(
                         _(
                             b'marking source repository as being upgraded; clients will be '
                             b'unable to read from repository\n'
                         )
                     )
                     scmutil.writereporequirements(
                         srcrepo, srcrepo.requirements | {b'upgradeinprogress'}
                     )
                     ui.status(_(b'starting in-place swap of repository data\n'))
                     if upgrade_op.backup_store:
                         ui.status(
                             _(b'replaced files will be backed up at %s\n') % backuppath
                         )
                     # Now swap in the new store directory. Doing it as a rename should make
                     # the operation nearly instantaneous and atomic (at least in well-behaved
                     # environments).
                     ui.status(_(b'replacing store...\n'))
                     tstart = util.timer()
                     _replacestores(srcrepo, dstrepo, backupvfs, upgrade_op)
                     elapsed = util.timer() - tstart
                     ui.status(
                         _(
                             b'store replacement complete; repository was inconsistent for '
                             b'%0.1fs\n'
                         )
                         % elapsed
                     )
                     # We first write the requirements file. Any new requirements will lock
                     # out legacy clients.
                     ui.status(
                         _(
                             b'finalizing requirements file and making repository readable '
                             b'again\n'
                         )
                     )
                     scmutil.writereporequirements(srcrepo, upgrade_op.new_requirements)
                     if upgrade_op.backup_store:
                         # The lock file from the old store won't be removed because nothing has a
                         # reference to its new location. So clean it up manually. Alternatively, we
                         # could update srcrepo.svfs and other variables to point to the new
                         # location. This is simpler.
                         assert backupvfs is not None  # help pytype
                         backupvfs.unlink(b'store/lock')
                 return backuppath

mercurial/verify.py

0 +2 -2

             # verify.py - repository integrity checking for Mercurial
             #
             # Copyright 2006, 2007 Olivia Mackall <olivia@selenic.com>
             #
             # This software may be used and distributed according to the terms of the
             # GNU General Public License version 2 or any later version.
             from __future__ import absolute_import
             import os
             from .i18n import _
             from .node import (
                 nullid,
                 short,
             )
             from .utils import (
                 stringutil,
             )
             from . import (
                 error,
                 pycompat,
                 revlog,
                 util,
             )
             VERIFY_DEFAULT = 0
             VERIFY_FULL = 1
             def verify(repo, level=None):
                 with repo.lock():
                     v = verifier(repo, level)
                     return v.verify()
             def _normpath(f):
                 # under hg < 2.4, convert didn't sanitize paths properly, so a
                 # converted repo may contain repeated slashes
                 while b'//' in f:
                     f = f.replace(b'//', b'/')
                 return f
             class verifier(object):
                 def __init__(self, repo, level=None):
                     self.repo = repo.unfiltered()
                     self.ui = repo.ui
                     self.match = repo.narrowmatch()
                     if level is None:
                         level = VERIFY_DEFAULT
                     self._level = level
                     self.badrevs = set()
                     self.errors = 0
                     self.warnings = 0
                     self.havecl = len(repo.changelog) > 0
                     self.havemf = len(repo.manifestlog.getstorage(b'')) > 0
                     self.revlogv1 = repo.changelog.version != revlog.REVLOGV0
                     self.lrugetctx = util.lrucachefunc(repo.unfiltered().__getitem__)
                     self.refersmf = False
                     self.fncachewarned = False
                     # developer config: verify.skipflags
                     self.skipflags = repo.ui.configint(b'verify', b'skipflags')
                     self.warnorphanstorefiles = True
                 def _warn(self, msg):
                     """record a "warning" level issue"""
                     self.ui.warn(msg + b"\n")
                     self.warnings += 1
                 def _err(self, linkrev, msg, filename=None):
                     """record a "error" level issue"""
                     if linkrev is not None:
                         self.badrevs.add(linkrev)
                         linkrev = b"%d" % linkrev
                     else:
                         linkrev = b'?'
                     msg = b"%s: %s" % (linkrev, msg)
                     if filename:
                         msg = b"%s@%s" % (filename, msg)
                     self.ui.warn(b" " + msg + b"\n")
                     self.errors += 1
                 def _exc(self, linkrev, msg, inst, filename=None):
                     """record exception raised during the verify process"""
                     fmsg = stringutil.forcebytestr(inst)
                     if not fmsg:
                         fmsg = pycompat.byterepr(inst)
                     self._err(linkrev, b"%s: %s" % (msg, fmsg), filename)
                 def _checkrevlog(self, obj, name, linkrev):
                     """verify high level property of a revlog
                     - revlog is present,
                     - revlog is non-empty,
                     - sizes (index and data) are correct,
                     - revlog's format version is correct.
                     """
                     if not len(obj) and (self.havecl or self.havemf):
                         self._err(linkrev, _(b"empty or missing %s") % name)
                         return
                     d = obj.checksize()
                     if d[0]:
                         self._err(None, _(b"data length off by %d bytes") % d[0], name)
                     if d[1]:
                         self._err(None, _(b"index contains %d extra bytes") % d[1], name)
                     if obj.version != revlog.REVLOGV0:
                         if not self.revlogv1:
                             self._warn(_(b"warning: `%s' uses revlog format 1") % name)
                     elif self.revlogv1:
                         self._warn(_(b"warning: `%s' uses revlog format 0") % name)
                 def _checkentry(self, obj, i, node, seen, linkrevs, f):
                     """verify a single revlog entry
                     arguments are:
                     - obj:      the source revlog
                     - i:        the revision number
                     - node:        the revision node id
                     - seen:     nodes previously seen for this revlog
                     - linkrevs: [changelog-revisions] introducing "node"
                     - f:        string label ("changelog", "manifest", or filename)
                     Performs the following checks:
                     - linkrev points to an existing changelog revision,
                     - linkrev points to a changelog revision that introduces this revision,
                     - linkrev points to the lowest of these changesets,
                     - both parents exist in the revlog,
                     - the revision is not duplicated.
                     Return the linkrev of the revision (or None for changelog's revisions).
                     """
                     lr = obj.linkrev(obj.rev(node))
                     if lr < 0 or (self.havecl and lr not in linkrevs):
                         if lr < 0 or lr >= len(self.repo.changelog):
                             msg = _(b"rev %d points to nonexistent changeset %d")
                         else:
                             msg = _(b"rev %d points to unexpected changeset %d")
                         self._err(None, msg % (i, lr), f)
                         if linkrevs:
                             if f and len(linkrevs) > 1:
                                 try:
                                     # attempt to filter down to real linkrevs
                                     linkrevs = [
                                         l
                                         for l in linkrevs
                                         if self.lrugetctx(l)[f].filenode() == node
                                     ]
                                 except Exception:
                                     pass
                             self._warn(
                                 _(b" (expected %s)")
                                 % b" ".join(map(pycompat.bytestr, linkrevs))
                             )
                         lr = None  # can't be trusted
                     try:
                         p1, p2 = obj.parents(node)
                         if p1 not in seen and p1 != nullid:
                             self._err(
                                 lr,
                                 _(b"unknown parent 1 %s of %s") % (short(p1), short(node)),
                                 f,
                             )
                         if p2 not in seen and p2 != nullid:
                             self._err(
                                 lr,
                                 _(b"unknown parent 2 %s of %s") % (short(p2), short(node)),
                                 f,
                             )
                     except Exception as inst:
                         self._exc(lr, _(b"checking parents of %s") % short(node), inst, f)
                     if node in seen:
                         self._err(lr, _(b"duplicate revision %d (%d)") % (i, seen[node]), f)
                     seen[node] = i
                     return lr
                 def verify(self):
                     """verify the content of the Mercurial repository
                     This method run all verifications, displaying issues as they are found.
                     return 1 if any error have been encountered, 0 otherwise."""
                     # initial validation and generic report
                     repo = self.repo
                     ui = repo.ui
                     if not repo.url().startswith(b'file:'):
                         raise error.Abort(_(b"cannot verify bundle or remote repos"))
                     if os.path.exists(repo.sjoin(b"journal")):
                         ui.warn(_(b"abandoned transaction found - run hg recover\n"))
                     if ui.verbose or not self.revlogv1:
                         ui.status(
                             _(b"repository uses revlog format %d\n")
                             % (self.revlogv1 and 1 or 0)
                         )
                     # data verification
                     mflinkrevs, filelinkrevs = self._verifychangelog()
                     filenodes = self._verifymanifest(mflinkrevs)
                     del mflinkrevs
                     self._crosscheckfiles(filelinkrevs, filenodes)
                     totalfiles, filerevisions = self._verifyfiles(filenodes, filelinkrevs)
                     # final report
                     ui.status(
                         _(b"checked %d changesets with %d changes to %d files\n")
                         % (len(repo.changelog), filerevisions, totalfiles)
                     )
                     if self.warnings:
                         ui.warn(_(b"%d warnings encountered!\n") % self.warnings)
                     if self.fncachewarned:
                         ui.warn(
                             _(
                                 b'hint: run "hg debugrebuildfncache" to recover from '
                                 b'corrupt fncache\n'
                             )
                         )
                     if self.errors:
                         ui.warn(_(b"%d integrity errors encountered!\n") % self.errors)
                         if self.badrevs:
                             ui.warn(
                                 _(b"(first damaged changeset appears to be %d)\n")
                                 % min(self.badrevs)
                             )
                         return 1
                     return 0
                 def _verifychangelog(self):
                     """verify the changelog of a repository
                     The following checks are performed:
                     - all of `_checkrevlog` checks,
                     - all of `_checkentry` checks (for each revisions),
                     - each revision can be read.
                     The function returns some of the data observed in the changesets as a
                     (mflinkrevs, filelinkrevs) tuples:
                     - mflinkrevs:   is a { manifest-node -> [changelog-rev] } mapping
                     - filelinkrevs: is a { file-path -> [changelog-rev] } mapping
                     If a matcher was specified, filelinkrevs will only contains matched
                     files.
                     """
                     ui = self.ui
                     repo = self.repo
                     match = self.match
                     cl = repo.changelog
                     ui.status(_(b"checking changesets\n"))
                     mflinkrevs = {}
                     filelinkrevs = {}
                     seen = {}
                     self._checkrevlog(cl, b"changelog", 0)
                     progress = ui.makeprogress(
                         _(b'checking'), unit=_(b'changesets'), total=len(repo)
                     )
                     for i in repo:
                         progress.update(i)
                         n = cl.node(i)
                         self._checkentry(cl, i, n, seen, [i], b"changelog")
                         try:
                             changes = cl.read(n)
                             if changes[0] != nullid:
                                 mflinkrevs.setdefault(changes[0], []).append(i)
                                 self.refersmf = True
                             for f in changes[3]:
                                 if match(f):
                                     filelinkrevs.setdefault(_normpath(f), []).append(i)
                         except Exception as inst:
                             self.refersmf = True
                             self._exc(i, _(b"unpacking changeset %s") % short(n), inst)
                     progress.complete()
                     return mflinkrevs, filelinkrevs
                 def _verifymanifest(
                     self, mflinkrevs, dir=b"", storefiles=None, subdirprogress=None
                 ):
                     """verify the manifestlog content
                     Inputs:
                     - mflinkrevs:     a {manifest-node -> [changelog-revisions]} mapping
                     - dir:            a subdirectory to check (for tree manifest repo)
                     - storefiles:     set of currently "orphan" files.
                     - subdirprogress: a progress object
                     This function checks:
                     * all of `_checkrevlog` checks (for all manifest related revlogs)
                     * all of `_checkentry` checks (for all manifest related revisions)
                     * nodes for subdirectory exists in the sub-directory manifest
                     * each manifest entries have a file path
                     * each manifest node refered in mflinkrevs exist in the manifest log
                     If tree manifest is in use and a matchers is specified, only the
                     sub-directories matching it will be verified.
                     return a two level mapping:
                         {"path" -> { filenode -> changelog-revision}}
                     This mapping primarily contains entries for every files in the
                     repository. In addition, when tree-manifest is used, it also contains
                     sub-directory entries.
                     If a matcher is provided, only matching paths will be included.
                     """
                     repo = self.repo
                     ui = self.ui
                     match = self.match
                     mfl = self.repo.manifestlog
                     mf = mfl.getstorage(dir)
                     if not dir:
                         self.ui.status(_(b"checking manifests\n"))
                     filenodes = {}
                     subdirnodes = {}
                     seen = {}
                     label = b"manifest"
                     if dir:
                         label = dir
                         revlogfiles = mf.files()
                         storefiles.difference_update(revlogfiles)
                         if subdirprogress:  # should be true since we're in a subdirectory
                             subdirprogress.increment()
                     if self.refersmf:
                         # Do not check manifest if there are only changelog entries with
                         # null manifests.
                         self._checkrevlog(mf, label, 0)
                     progress = ui.makeprogress(
                         _(b'checking'), unit=_(b'manifests'), total=len(mf)
                     )
                     for i in mf:
                         if not dir:
                             progress.update(i)
                         n = mf.node(i)
                         lr = self._checkentry(mf, i, n, seen, mflinkrevs.get(n, []), label)
                         if n in mflinkrevs:
                             del mflinkrevs[n]
                         elif dir:
                             self._err(
                                 lr,
                                 _(b"%s not in parent-directory manifest") % short(n),
                                 label,
                             )
                         else:
                             self._err(lr, _(b"%s not in changesets") % short(n), label)
                         try:
                             mfdelta = mfl.get(dir, n).readdelta(shallow=True)
                             for f, fn, fl in mfdelta.iterentries():
                                 if not f:
                                     self._err(lr, _(b"entry without name in manifest"))
                                 elif f == b"/dev/null":  # ignore this in very old repos
                                     continue
                                 fullpath = dir + _normpath(f)
                                 if fl == b't':
                                     if not match.visitdir(fullpath):
                                         continue
                                     subdirnodes.setdefault(fullpath + b'/', {}).setdefault(
                                         fn, []
                                     ).append(lr)
                                 else:
                                     if not match(fullpath):
                                         continue
                                     filenodes.setdefault(fullpath, {}).setdefault(fn, lr)
                         except Exception as inst:
                             self._exc(lr, _(b"reading delta %s") % short(n), inst, label)
                         if self._level >= VERIFY_FULL:
                             try:
                                 # Various issues can affect manifest. So we read each full
                                 # text from storage. This triggers the checks from the core
                                 # code (eg: hash verification, filename are ordered, etc.)
                                 mfdelta = mfl.get(dir, n).read()
                             except Exception as inst:
                                 self._exc(
                                     lr,
                                     _(b"reading full manifest %s") % short(n),
                                     inst,
                                     label,
                                 )
                     if not dir:
                         progress.complete()
                     if self.havemf:
                         # since we delete entry in `mflinkrevs` during iteration, any
                         # remaining entries are "missing". We need to issue errors for them.
                         changesetpairs = [(c, m) for m in mflinkrevs for c in mflinkrevs[m]]
                         for c, m in sorted(changesetpairs):
                             if dir:
                                 self._err(
                                     c,
                                     _(
                                         b"parent-directory manifest refers to unknown"
                                         b" revision %s"
                                     )
                                     % short(m),
                                     label,
                                 )
                             else:
                                 self._err(
                                     c,
                                     _(b"changeset refers to unknown revision %s")
                                     % short(m),
                                     label,
                                 )
                     if not dir and subdirnodes:
                         self.ui.status(_(b"checking directory manifests\n"))
                         storefiles = set()
                         subdirs = set()
                         revlogv1 = self.revlogv1
-                        for f, f2, size in repo.store.datafiles():
+                        for t, f, f2, size in repo.store.datafiles():
                             if not f:
                                 self._err(None, _(b"cannot decode filename '%s'") % f2)
                             elif (size > 0 or not revlogv1) and f.startswith(b'meta/'):
                                 storefiles.add(_normpath(f))
                                 subdirs.add(os.path.dirname(f))
                         subdirprogress = ui.makeprogress(
                             _(b'checking'), unit=_(b'manifests'), total=len(subdirs)
                         )
                     for subdir, linkrevs in pycompat.iteritems(subdirnodes):
                         subdirfilenodes = self._verifymanifest(
                             linkrevs, subdir, storefiles, subdirprogress
                         )
                         for f, onefilenodes in pycompat.iteritems(subdirfilenodes):
                             filenodes.setdefault(f, {}).update(onefilenodes)
                     if not dir and subdirnodes:
                         assert subdirprogress is not None  # help pytype
                         subdirprogress.complete()
                         if self.warnorphanstorefiles:
                             for f in sorted(storefiles):
                                 self._warn(_(b"warning: orphan data file '%s'") % f)
                     return filenodes
                 def _crosscheckfiles(self, filelinkrevs, filenodes):
                     repo = self.repo
                     ui = self.ui
                     ui.status(_(b"crosschecking files in changesets and manifests\n"))
                     total = len(filelinkrevs) + len(filenodes)
                     progress = ui.makeprogress(
                         _(b'crosschecking'), unit=_(b'files'), total=total
                     )
                     if self.havemf:
                         for f in sorted(filelinkrevs):
                             progress.increment()
                             if f not in filenodes:
                                 lr = filelinkrevs[f][0]
                                 self._err(lr, _(b"in changeset but not in manifest"), f)
                     if self.havecl:
                         for f in sorted(filenodes):
                             progress.increment()
                             if f not in filelinkrevs:
                                 try:
                                     fl = repo.file(f)
                                     lr = min([fl.linkrev(fl.rev(n)) for n in filenodes[f]])
                                 except Exception:
                                     lr = None
                                 self._err(lr, _(b"in manifest but not in changeset"), f)
                     progress.complete()
                 def _verifyfiles(self, filenodes, filelinkrevs):
                     repo = self.repo
                     ui = self.ui
                     lrugetctx = self.lrugetctx
                     revlogv1 = self.revlogv1
                     havemf = self.havemf
                     ui.status(_(b"checking files\n"))
                     storefiles = set()
-                    for f, f2, size in repo.store.datafiles():
+                    for rl_type, f, f2, size in repo.store.datafiles():
                         if not f:
                             self._err(None, _(b"cannot decode filename '%s'") % f2)
                         elif (size > 0 or not revlogv1) and f.startswith(b'data/'):
                             storefiles.add(_normpath(f))
                     state = {
                         # TODO this assumes revlog storage for changelog.
                         b'expectedversion': self.repo.changelog.version & 0xFFFF,
                         b'skipflags': self.skipflags,
                         # experimental config: censor.policy
                         b'erroroncensored': ui.config(b'censor', b'policy') == b'abort',
                     }
                     files = sorted(set(filenodes) | set(filelinkrevs))
                     revisions = 0
                     progress = ui.makeprogress(
                         _(b'checking'), unit=_(b'files'), total=len(files)
                     )
                     for i, f in enumerate(files):
                         progress.update(i, item=f)
                         try:
                             linkrevs = filelinkrevs[f]
                         except KeyError:
                             # in manifest but not in changelog
                             linkrevs = []
                         if linkrevs:
                             lr = linkrevs[0]
                         else:
                             lr = None
                         try:
                             fl = repo.file(f)
                         except error.StorageError as e:
                             self._err(lr, _(b"broken revlog! (%s)") % e, f)
                             continue
                         for ff in fl.files():
                             try:
                                 storefiles.remove(ff)
                             except KeyError:
                                 if self.warnorphanstorefiles:
                                     self._warn(
                                         _(b" warning: revlog '%s' not in fncache!") % ff
                                     )
                                     self.fncachewarned = True
                         if not len(fl) and (self.havecl or self.havemf):
                             self._err(lr, _(b"empty or missing %s") % f)
                         else:
                             # Guard against implementations not setting this.
                             state[b'skipread'] = set()
                             state[b'safe_renamed'] = set()
                             for problem in fl.verifyintegrity(state):
                                 if problem.node is not None:
                                     linkrev = fl.linkrev(fl.rev(problem.node))
                                 else:
                                     linkrev = None
                                 if problem.warning:
                                     self._warn(problem.warning)
                                 elif problem.error:
                                     self._err(
                                         linkrev if linkrev is not None else lr,
                                         problem.error,
                                         f,
                                     )
                                 else:
                                     raise error.ProgrammingError(
                                         b'problem instance does not set warning or error '
                                         b'attribute: %s' % problem.msg
                                     )
                         seen = {}
                         for i in fl:
                             revisions += 1
                             n = fl.node(i)
                             lr = self._checkentry(fl, i, n, seen, linkrevs, f)
                             if f in filenodes:
                                 if havemf and n not in filenodes[f]:
                                     self._err(lr, _(b"%s not in manifests") % (short(n)), f)
                                 else:
                                     del filenodes[f][n]
                             if n in state[b'skipread'] and n not in state[b'safe_renamed']:
                                 continue
                             # check renames
                             try:
                                 # This requires resolving fulltext (at least on revlogs,
                                 # though not with LFS revisions). We may want
                                 # ``verifyintegrity()`` to pass a set of nodes with
                                 # rename metadata as an optimization.
                                 rp = fl.renamed(n)
                                 if rp:
                                     if lr is not None and ui.verbose:
                                         ctx = lrugetctx(lr)
                                         if not any(rp[0] in pctx for pctx in ctx.parents()):
                                             self._warn(
                                                 _(
                                                     b"warning: copy source of '%s' not"
                                                     b" in parents of %s"
                                                 )
                                                 % (f, ctx)
                                             )
                                     fl2 = repo.file(rp[0])
                                     if not len(fl2):
                                         self._err(
                                             lr,
                                             _(
                                                 b"empty or missing copy source revlog "
                                                 b"%s:%s"
                                             )
                                             % (rp[0], short(rp[1])),
                                             f,
                                         )
                                     elif rp[1] == nullid:
                                         ui.note(
                                             _(
                                                 b"warning: %s@%s: copy source"
                                                 b" revision is nullid %s:%s\n"
                                             )
                                             % (f, lr, rp[0], short(rp[1]))
                                         )
                                     else:
                                         fl2.rev(rp[1])
                             except Exception as inst:
                                 self._exc(
                                     lr, _(b"checking rename of %s") % short(n), inst, f
                                 )
                         # cross-check
                         if f in filenodes:
                             fns = [(v, k) for k, v in pycompat.iteritems(filenodes[f])]
                             for lr, node in sorted(fns):
                                 self._err(
                                     lr,
                                     _(b"manifest refers to unknown revision %s")
                                     % short(node),
                                     f,
                                 )
                     progress.complete()
                     if self.warnorphanstorefiles:
                         for f in sorted(storefiles):
                             self._warn(_(b"warning: orphan data file '%s'") % f)
                     return len(files), revisions

mercurial/wireprotov2server.py

0 +2 -1

             # Copyright 21 May 2005 - (c) 2005 Jake Edge <jake@edge2.net>
             # Copyright 2005-2007 Olivia Mackall <olivia@selenic.com>
             #
             # This software may be used and distributed according to the terms of the
             # GNU General Public License version 2 or any later version.
             from __future__ import absolute_import
             import collections
             import contextlib
             from .i18n import _
             from .node import (
                 hex,
                 nullid,
             )
             from . import (
                 discovery,
                 encoding,
                 error,
                 match as matchmod,
                 narrowspec,
                 pycompat,
                 streamclone,
                 templatefilters,
                 util,
                 wireprotoframing,
                 wireprototypes,
             )
             from .interfaces import util as interfaceutil
             from .utils import (
                 cborutil,
                 hashutil,
                 stringutil,
             )
             FRAMINGTYPE = b'application/mercurial-exp-framing-0006'
             HTTP_WIREPROTO_V2 = wireprototypes.HTTP_WIREPROTO_V2
             COMMANDS = wireprototypes.commanddict()
             # Value inserted into cache key computation function. Change the value to
             # force new cache keys for every command request. This should be done when
             # there is a change to how caching works, etc.
             GLOBAL_CACHE_VERSION = 1
             def handlehttpv2request(rctx, req, res, checkperm, urlparts):
                 from .hgweb import common as hgwebcommon
                 # URL space looks like: <permissions>/<command>, where <permission> can
                 # be ``ro`` or ``rw`` to signal read-only or read-write, respectively.
                 # Root URL does nothing meaningful... yet.
                 if not urlparts:
                     res.status = b'200 OK'
                     res.headers[b'Content-Type'] = b'text/plain'
                     res.setbodybytes(_(b'HTTP version 2 API handler'))
                     return
                 if len(urlparts) == 1:
                     res.status = b'404 Not Found'
                     res.headers[b'Content-Type'] = b'text/plain'
                     res.setbodybytes(
                         _(b'do not know how to process %s\n') % req.dispatchpath
                     )
                     return
                 permission, command = urlparts[0:2]
                 if permission not in (b'ro', b'rw'):
                     res.status = b'404 Not Found'
                     res.headers[b'Content-Type'] = b'text/plain'
                     res.setbodybytes(_(b'unknown permission: %s') % permission)
                     return
                 if req.method != b'POST':
                     res.status = b'405 Method Not Allowed'
                     res.headers[b'Allow'] = b'POST'
                     res.setbodybytes(_(b'commands require POST requests'))
                     return
                 # At some point we'll want to use our own API instead of recycling the
                 # behavior of version 1 of the wire protocol...
                 # TODO return reasonable responses - not responses that overload the
                 # HTTP status line message for error reporting.
                 try:
                     checkperm(rctx, req, b'pull' if permission == b'ro' else b'push')
                 except hgwebcommon.ErrorResponse as e:
                     res.status = hgwebcommon.statusmessage(
                         e.code, stringutil.forcebytestr(e)
                     )
                     for k, v in e.headers:
                         res.headers[k] = v
                     res.setbodybytes(b'permission denied')
                     return
                 # We have a special endpoint to reflect the request back at the client.
                 if command == b'debugreflect':
                     _processhttpv2reflectrequest(rctx.repo.ui, rctx.repo, req, res)
                     return
                 # Extra commands that we handle that aren't really wire protocol
                 # commands. Think extra hard before making this hackery available to
                 # extension.
                 extracommands = {b'multirequest'}
                 if command not in COMMANDS and command not in extracommands:
                     res.status = b'404 Not Found'
                     res.headers[b'Content-Type'] = b'text/plain'
                     res.setbodybytes(_(b'unknown wire protocol command: %s\n') % command)
                     return
                 repo = rctx.repo
                 ui = repo.ui
                 proto = httpv2protocolhandler(req, ui)
                 if (
                     not COMMANDS.commandavailable(command, proto)
                     and command not in extracommands
                 ):
                     res.status = b'404 Not Found'
                     res.headers[b'Content-Type'] = b'text/plain'
                     res.setbodybytes(_(b'invalid wire protocol command: %s') % command)
                     return
                 # TODO consider cases where proxies may add additional Accept headers.
                 if req.headers.get(b'Accept') != FRAMINGTYPE:
                     res.status = b'406 Not Acceptable'
                     res.headers[b'Content-Type'] = b'text/plain'
                     res.setbodybytes(
                         _(b'client MUST specify Accept header with value: %s\n')
                         % FRAMINGTYPE
                     )
                     return
                 if req.headers.get(b'Content-Type') != FRAMINGTYPE:
                     res.status = b'415 Unsupported Media Type'
                     # TODO we should send a response with appropriate media type,
                     # since client does Accept it.
                     res.headers[b'Content-Type'] = b'text/plain'
                     res.setbodybytes(
                         _(b'client MUST send Content-Type header with value: %s\n')
                         % FRAMINGTYPE
                     )
                     return
                 _processhttpv2request(ui, repo, req, res, permission, command, proto)
             def _processhttpv2reflectrequest(ui, repo, req, res):
                 """Reads unified frame protocol request and dumps out state to client.
                 This special endpoint can be used to help debug the wire protocol.
                 Instead of routing the request through the normal dispatch mechanism,
                 we instead read all frames, decode them, and feed them into our state
                 tracker. We then dump the log of all that activity back out to the
                 client.
                 """
                 # Reflection APIs have a history of being abused, accidentally disclosing
                 # sensitive data, etc. So we have a config knob.
                 if not ui.configbool(b'experimental', b'web.api.debugreflect'):
                     res.status = b'404 Not Found'
                     res.headers[b'Content-Type'] = b'text/plain'
                     res.setbodybytes(_(b'debugreflect service not available'))
                     return
                 # We assume we have a unified framing protocol request body.
                 reactor = wireprotoframing.serverreactor(ui)
                 states = []
                 while True:
                     frame = wireprotoframing.readframe(req.bodyfh)
                     if not frame:
                         states.append(b'received: <no frame>')
                         break
                     states.append(
                         b'received: %d %d %d %s'
                         % (frame.typeid, frame.flags, frame.requestid, frame.payload)
                     )
                     action, meta = reactor.onframerecv(frame)
                     states.append(templatefilters.json((action, meta)))
                 action, meta = reactor.oninputeof()
                 meta[b'action'] = action
                 states.append(templatefilters.json(meta))
                 res.status = b'200 OK'
                 res.headers[b'Content-Type'] = b'text/plain'
                 res.setbodybytes(b'\n'.join(states))
             def _processhttpv2request(ui, repo, req, res, authedperm, reqcommand, proto):
                 """Post-validation handler for HTTPv2 requests.
                 Called when the HTTP request contains unified frame-based protocol
                 frames for evaluation.
                 """
                 # TODO Some HTTP clients are full duplex and can receive data before
                 # the entire request is transmitted. Figure out a way to indicate support
                 # for that so we can opt into full duplex mode.
                 reactor = wireprotoframing.serverreactor(ui, deferoutput=True)
                 seencommand = False
                 outstream = None
                 while True:
                     frame = wireprotoframing.readframe(req.bodyfh)
                     if not frame:
                         break
                     action, meta = reactor.onframerecv(frame)
                     if action == b'wantframe':
                         # Need more data before we can do anything.
                         continue
                     elif action == b'runcommand':
                         # Defer creating output stream because we need to wait for
                         # protocol settings frames so proper encoding can be applied.
                         if not outstream:
                             outstream = reactor.makeoutputstream()
                         sentoutput = _httpv2runcommand(
                             ui,
                             repo,
                             req,
                             res,
                             authedperm,
                             reqcommand,
                             reactor,
                             outstream,
                             meta,
                             issubsequent=seencommand,
                         )
                         if sentoutput:
                             return
                         seencommand = True
                     elif action == b'error':
                         # TODO define proper error mechanism.
                         res.status = b'200 OK'
                         res.headers[b'Content-Type'] = b'text/plain'
                         res.setbodybytes(meta[b'message'] + b'\n')
                         return
                     else:
                         raise error.ProgrammingError(
                             b'unhandled action from frame processor: %s' % action
                         )
                 action, meta = reactor.oninputeof()
                 if action == b'sendframes':
                     # We assume we haven't started sending the response yet. If we're
                     # wrong, the response type will raise an exception.
                     res.status = b'200 OK'
                     res.headers[b'Content-Type'] = FRAMINGTYPE
                     res.setbodygen(meta[b'framegen'])
                 elif action == b'noop':
                     pass
                 else:
                     raise error.ProgrammingError(
                         b'unhandled action from frame processor: %s' % action
                     )
             def _httpv2runcommand(
                 ui,
                 repo,
                 req,
                 res,
                 authedperm,
                 reqcommand,
                 reactor,
                 outstream,
                 command,
                 issubsequent,
             ):
                 """Dispatch a wire protocol command made from HTTPv2 requests.
                 The authenticated permission (``authedperm``) along with the original
                 command from the URL (``reqcommand``) are passed in.
                 """
                 # We already validated that the session has permissions to perform the
                 # actions in ``authedperm``. In the unified frame protocol, the canonical
                 # command to run is expressed in a frame. However, the URL also requested
                 # to run a specific command. We need to be careful that the command we
                 # run doesn't have permissions requirements greater than what was granted
                 # by ``authedperm``.
                 #
                 # Our rule for this is we only allow one command per HTTP request and
                 # that command must match the command in the URL. However, we make
                 # an exception for the ``multirequest`` URL. This URL is allowed to
                 # execute multiple commands. We double check permissions of each command
                 # as it is invoked to ensure there is no privilege escalation.
                 # TODO consider allowing multiple commands to regular command URLs
                 # iff each command is the same.
                 proto = httpv2protocolhandler(req, ui, args=command[b'args'])
                 if reqcommand == b'multirequest':
                     if not COMMANDS.commandavailable(command[b'command'], proto):
                         # TODO proper error mechanism
                         res.status = b'200 OK'
                         res.headers[b'Content-Type'] = b'text/plain'
                         res.setbodybytes(
                             _(b'wire protocol command not available: %s')
                             % command[b'command']
                         )
                         return True
                     # TODO don't use assert here, since it may be elided by -O.
                     assert authedperm in (b'ro', b'rw')
                     wirecommand = COMMANDS[command[b'command']]
                     assert wirecommand.permission in (b'push', b'pull')
                     if authedperm == b'ro' and wirecommand.permission != b'pull':
                         # TODO proper error mechanism
                         res.status = b'403 Forbidden'
                         res.headers[b'Content-Type'] = b'text/plain'
                         res.setbodybytes(
                             _(b'insufficient permissions to execute command: %s')
                             % command[b'command']
                         )
                         return True
                     # TODO should we also call checkperm() here? Maybe not if we're going
                     # to overhaul that API. The granted scope from the URL check should
                     # be good enough.
                 else:
                     # Don't allow multiple commands outside of ``multirequest`` URL.
                     if issubsequent:
                         # TODO proper error mechanism
                         res.status = b'200 OK'
                         res.headers[b'Content-Type'] = b'text/plain'
                         res.setbodybytes(
                             _(b'multiple commands cannot be issued to this URL')
                         )
                         return True
                     if reqcommand != command[b'command']:
                         # TODO define proper error mechanism
                         res.status = b'200 OK'
                         res.headers[b'Content-Type'] = b'text/plain'
                         res.setbodybytes(_(b'command in frame must match command in URL'))
                         return True
                 res.status = b'200 OK'
                 res.headers[b'Content-Type'] = FRAMINGTYPE
                 try:
                     objs = dispatch(repo, proto, command[b'command'], command[b'redirect'])
                     action, meta = reactor.oncommandresponsereadyobjects(
                         outstream, command[b'requestid'], objs
                     )
                 except error.WireprotoCommandError as e:
                     action, meta = reactor.oncommanderror(
                         outstream, command[b'requestid'], e.message, e.messageargs
                     )
                 except Exception as e:
                     action, meta = reactor.onservererror(
                         outstream,
                         command[b'requestid'],
                         _(b'exception when invoking command: %s')
                         % stringutil.forcebytestr(e),
                     )
                 if action == b'sendframes':
                     res.setbodygen(meta[b'framegen'])
                     return True
                 elif action == b'noop':
                     return False
                 else:
                     raise error.ProgrammingError(
                         b'unhandled event from reactor: %s' % action
                     )
             def getdispatchrepo(repo, proto, command):
                 viewconfig = repo.ui.config(b'server', b'view')
                 return repo.filtered(viewconfig)
             def dispatch(repo, proto, command, redirect):
                 """Run a wire protocol command.
                 Returns an iterable of objects that will be sent to the client.
                 """
                 repo = getdispatchrepo(repo, proto, command)
                 entry = COMMANDS[command]
                 func = entry.func
                 spec = entry.args
                 args = proto.getargs(spec)
                 # There is some duplicate boilerplate code here for calling the command and
                 # emitting objects. It is either that or a lot of indented code that looks
                 # like a pyramid (since there are a lot of code paths that result in not
                 # using the cacher).
                 callcommand = lambda: func(repo, proto, **pycompat.strkwargs(args))
                 # Request is not cacheable. Don't bother instantiating a cacher.
                 if not entry.cachekeyfn:
                     for o in callcommand():
                         yield o
                     return
                 if redirect:
                     redirecttargets = redirect[b'targets']
                     redirecthashes = redirect[b'hashes']
                 else:
                     redirecttargets = []
                     redirecthashes = []
                 cacher = makeresponsecacher(
                     repo,
                     proto,
                     command,
                     args,
                     cborutil.streamencode,
                     redirecttargets=redirecttargets,
                     redirecthashes=redirecthashes,
                 )
                 # But we have no cacher. Do default handling.
                 if not cacher:
                     for o in callcommand():
                         yield o
                     return
                 with cacher:
                     cachekey = entry.cachekeyfn(
                         repo, proto, cacher, **pycompat.strkwargs(args)
                     )
                     # No cache key or the cacher doesn't like it. Do default handling.
                     if cachekey is None or not cacher.setcachekey(cachekey):
                         for o in callcommand():
                             yield o
                         return
                     # Serve it from the cache, if possible.
                     cached = cacher.lookup()
                     if cached:
                         for o in cached[b'objs']:
                             yield o
                         return
                     # Else call the command and feed its output into the cacher, allowing
                     # the cacher to buffer/mutate objects as it desires.
                     for o in callcommand():
                         for o in cacher.onobject(o):
                             yield o
                     for o in cacher.onfinished():
                         yield o
             @interfaceutil.implementer(wireprototypes.baseprotocolhandler)
             class httpv2protocolhandler(object):
                 def __init__(self, req, ui, args=None):
                     self._req = req
                     self._ui = ui
                     self._args = args
                 @property
                 def name(self):
                     return HTTP_WIREPROTO_V2
                 def getargs(self, args):
                     # First look for args that were passed but aren't registered on this
                     # command.
                     extra = set(self._args) - set(args)
                     if extra:
                         raise error.WireprotoCommandError(
                             b'unsupported argument to command: %s'
                             % b', '.join(sorted(extra))
                         )
                     # And look for required arguments that are missing.
                     missing = {a for a in args if args[a][b'required']} - set(self._args)
                     if missing:
                         raise error.WireprotoCommandError(
                             b'missing required arguments: %s' % b', '.join(sorted(missing))
                         )
                     # Now derive the arguments to pass to the command, taking into
                     # account the arguments specified by the client.
                     data = {}
                     for k, meta in sorted(args.items()):
                         # This argument wasn't passed by the client.
                         if k not in self._args:
                             data[k] = meta[b'default']()
                             continue
                         v = self._args[k]
                         # Sets may be expressed as lists. Silently normalize.
                         if meta[b'type'] == b'set' and isinstance(v, list):
                             v = set(v)
                         # TODO consider more/stronger type validation.
                         data[k] = v
                     return data
                 def getprotocaps(self):
                     # Protocol capabilities are currently not implemented for HTTP V2.
                     return set()
                 def getpayload(self):
                     raise NotImplementedError
                 @contextlib.contextmanager
                 def mayberedirectstdio(self):
                     raise NotImplementedError
                 def client(self):
                     raise NotImplementedError
                 def addcapabilities(self, repo, caps):
                     return caps
                 def checkperm(self, perm):
                     raise NotImplementedError
             def httpv2apidescriptor(req, repo):
                 proto = httpv2protocolhandler(req, repo.ui)
                 return _capabilitiesv2(repo, proto)
             def _capabilitiesv2(repo, proto):
                 """Obtain the set of capabilities for version 2 transports.
                 These capabilities are distinct from the capabilities for version 1
                 transports.
                 """
                 caps = {
                     b'commands': {},
                     b'framingmediatypes': [FRAMINGTYPE],
                     b'pathfilterprefixes': set(narrowspec.VALID_PREFIXES),
                 }
                 for command, entry in COMMANDS.items():
                     args = {}
                     for arg, meta in entry.args.items():
                         args[arg] = {
                             # TODO should this be a normalized type using CBOR's
                             # terminology?
                             b'type': meta[b'type'],
                             b'required': meta[b'required'],
                         }
                         if not meta[b'required']:
                             args[arg][b'default'] = meta[b'default']()
                         if meta[b'validvalues']:
                             args[arg][b'validvalues'] = meta[b'validvalues']
                     # TODO this type of check should be defined in a per-command callback.
                     if (
                         command == b'rawstorefiledata'
                         and not streamclone.allowservergeneration(repo)
                     ):
                         continue
                     caps[b'commands'][command] = {
                         b'args': args,
                         b'permissions': [entry.permission],
                     }
                     if entry.extracapabilitiesfn:
                         extracaps = entry.extracapabilitiesfn(repo, proto)
                         caps[b'commands'][command].update(extracaps)
                 caps[b'rawrepoformats'] = sorted(repo.requirements & repo.supportedformats)
                 targets = getadvertisedredirecttargets(repo, proto)
                 if targets:
                     caps[b'redirect'] = {
                         b'targets': [],
                         b'hashes': [b'sha256', b'sha1'],
                     }
                     for target in targets:
                         entry = {
                             b'name': target[b'name'],
                             b'protocol': target[b'protocol'],
                             b'uris': target[b'uris'],
                         }
                         for key in (b'snirequired', b'tlsversions'):
                             if key in target:
                                 entry[key] = target[key]
                         caps[b'redirect'][b'targets'].append(entry)
                 return proto.addcapabilities(repo, caps)
             def getadvertisedredirecttargets(repo, proto):
                 """Obtain a list of content redirect targets.
                 Returns a list containing potential redirect targets that will be
                 advertised in capabilities data. Each dict MUST have the following
                 keys:
                 name
                    The name of this redirect target. This is the identifier clients use
                    to refer to a target. It is transferred as part of every command
                    request.
                 protocol
                    Network protocol used by this target. Typically this is the string
                    in front of the ``://`` in a URL. e.g. ``https``.
                 uris
                    List of representative URIs for this target. Clients can use the
                    URIs to test parsing for compatibility or for ordering preference
                    for which target to use.
                 The following optional keys are recognized:
                 snirequired
                    Bool indicating if Server Name Indication (SNI) is required to
                    connect to this target.
                 tlsversions
                    List of bytes indicating which TLS versions are supported by this
                    target.
                 By default, clients reflect the target order advertised by servers
                 and servers will use the first client-advertised target when picking
                 a redirect target. So targets should be advertised in the order the
                 server prefers they be used.
                 """
                 return []
             def wireprotocommand(
                 name,
                 args=None,
                 permission=b'push',
                 cachekeyfn=None,
                 extracapabilitiesfn=None,
             ):
                 """Decorator to declare a wire protocol command.
                 ``name`` is the name of the wire protocol command being provided.
                 ``args`` is a dict defining arguments accepted by the command. Keys are
                 the argument name. Values are dicts with the following keys:
                    ``type``
                       The argument data type. Must be one of the following string
                       literals: ``bytes``, ``int``, ``list``, ``dict``, ``set``,
                       or ``bool``.
                    ``default``
                       A callable returning the default value for this argument. If not
                       specified, ``None`` will be the default value.
                    ``example``
                       An example value for this argument.
                    ``validvalues``
                       Set of recognized values for this argument.
                 ``permission`` defines the permission type needed to run this command.
                 Can be ``push`` or ``pull``. These roughly map to read-write and read-only,
                 respectively. Default is to assume command requires ``push`` permissions
                 because otherwise commands not declaring their permissions could modify
                 a repository that is supposed to be read-only.
                 ``cachekeyfn`` defines an optional callable that can derive the
                 cache key for this request.
                 ``extracapabilitiesfn`` defines an optional callable that defines extra
                 command capabilities/parameters that are advertised next to the command
                 in the capabilities data structure describing the server. The callable
                 receives as arguments the repository and protocol objects. It returns
                 a dict of extra fields to add to the command descriptor.
                 Wire protocol commands are generators of objects to be serialized and
                 sent to the client.
                 If a command raises an uncaught exception, this will be translated into
                 a command error.
                 All commands can opt in to being cacheable by defining a function
                 (``cachekeyfn``) that is called to derive a cache key. This function
                 receives the same arguments as the command itself plus a ``cacher``
                 argument containing the active cacher for the request and returns a bytes
                 containing the key in a cache the response to this command may be cached
                 under.
                 """
                 transports = {
                     k for k, v in wireprototypes.TRANSPORTS.items() if v[b'version'] == 2
                 }
                 if permission not in (b'push', b'pull'):
                     raise error.ProgrammingError(
                         b'invalid wire protocol permission; '
                         b'got %s; expected "push" or "pull"' % permission
                     )
                 if args is None:
                     args = {}
                 if not isinstance(args, dict):
                     raise error.ProgrammingError(
                         b'arguments for version 2 commands must be declared as dicts'
                     )
                 for arg, meta in args.items():
                     if arg == b'*':
                         raise error.ProgrammingError(
                             b'* argument name not allowed on version 2 commands'
                         )
                     if not isinstance(meta, dict):
                         raise error.ProgrammingError(
                             b'arguments for version 2 commands '
                             b'must declare metadata as a dict'
                         )
                     if b'type' not in meta:
                         raise error.ProgrammingError(
                             b'%s argument for command %s does not '
                             b'declare type field' % (arg, name)
                         )
                     if meta[b'type'] not in (
                         b'bytes',
                         b'int',
                         b'list',
                         b'dict',
                         b'set',
                         b'bool',
                     ):
                         raise error.ProgrammingError(
                             b'%s argument for command %s has '
                             b'illegal type: %s' % (arg, name, meta[b'type'])
                         )
                     if b'example' not in meta:
                         raise error.ProgrammingError(
                             b'%s argument for command %s does not '
                             b'declare example field' % (arg, name)
                         )
                     meta[b'required'] = b'default' not in meta
                     meta.setdefault(b'default', lambda: None)
                     meta.setdefault(b'validvalues', None)
                 def register(func):
                     if name in COMMANDS:
                         raise error.ProgrammingError(
                             b'%s command already registered for version 2' % name
                         )
                     COMMANDS[name] = wireprototypes.commandentry(
                         func,
                         args=args,
                         transports=transports,
                         permission=permission,
                         cachekeyfn=cachekeyfn,
                         extracapabilitiesfn=extracapabilitiesfn,
                     )
                     return func
                 return register
             def makecommandcachekeyfn(command, localversion=None, allargs=False):
                 """Construct a cache key derivation function with common features.
                 By default, the cache key is a hash of:
                 * The command name.
                 * A global cache version number.
                 * A local cache version number (passed via ``localversion``).
                 * All the arguments passed to the command.
                 * The media type used.
                 * Wire protocol version string.
                 * The repository path.
                 """
                 if not allargs:
                     raise error.ProgrammingError(
                         b'only allargs=True is currently supported'
                     )
                 if localversion is None:
                     raise error.ProgrammingError(b'must set localversion argument value')
                 def cachekeyfn(repo, proto, cacher, **args):
                     spec = COMMANDS[command]
                     # Commands that mutate the repo can not be cached.
                     if spec.permission == b'push':
                         return None
                     # TODO config option to disable caching.
                     # Our key derivation strategy is to construct a data structure
                     # holding everything that could influence cacheability and to hash
                     # the CBOR representation of that. Using CBOR seems like it might
                     # be overkill. However, simpler hashing mechanisms are prone to
                     # duplicate input issues. e.g. if you just concatenate two values,
                     # "foo"+"bar" is identical to "fo"+"obar". Using CBOR provides
                     # "padding" between values and prevents these problems.
                     # Seed the hash with various data.
                     state = {
                         # To invalidate all cache keys.
                         b'globalversion': GLOBAL_CACHE_VERSION,
                         # More granular cache key invalidation.
                         b'localversion': localversion,
                         # Cache keys are segmented by command.
                         b'command': command,
                         # Throw in the media type and API version strings so changes
                         # to exchange semantics invalid cache.
                         b'mediatype': FRAMINGTYPE,
                         b'version': HTTP_WIREPROTO_V2,
                         # So same requests for different repos don't share cache keys.
                         b'repo': repo.root,
                     }
                     # The arguments passed to us will have already been normalized.
                     # Default values will be set, etc. This is important because it
                     # means that it doesn't matter if clients send an explicit argument
                     # or rely on the default value: it will all normalize to the same
                     # set of arguments on the server and therefore the same cache key.
                     #
                     # Arguments by their very nature must support being encoded to CBOR.
                     # And the CBOR encoder is deterministic. So we hash the arguments
                     # by feeding the CBOR of their representation into the hasher.
                     if allargs:
                         state[b'args'] = pycompat.byteskwargs(args)
                     cacher.adjustcachekeystate(state)
                     hasher = hashutil.sha1()
                     for chunk in cborutil.streamencode(state):
                         hasher.update(chunk)
                     return pycompat.sysbytes(hasher.hexdigest())
                 return cachekeyfn
             def makeresponsecacher(
                 repo, proto, command, args, objencoderfn, redirecttargets, redirecthashes
             ):
                 """Construct a cacher for a cacheable command.
                 Returns an ``iwireprotocolcommandcacher`` instance.
                 Extensions can monkeypatch this function to provide custom caching
                 backends.
                 """
                 return None
             def resolvenodes(repo, revisions):
                 """Resolve nodes from a revisions specifier data structure."""
                 cl = repo.changelog
                 clhasnode = cl.hasnode
                 seen = set()
                 nodes = []
                 if not isinstance(revisions, list):
                     raise error.WireprotoCommandError(
                         b'revisions must be defined as an array'
                     )
                 for spec in revisions:
                     if b'type' not in spec:
                         raise error.WireprotoCommandError(
                             b'type key not present in revision specifier'
                         )
                     typ = spec[b'type']
                     if typ == b'changesetexplicit':
                         if b'nodes' not in spec:
                             raise error.WireprotoCommandError(
                                 b'nodes key not present in changesetexplicit revision '
                                 b'specifier'
                             )
                         for node in spec[b'nodes']:
                             if node not in seen:
                                 nodes.append(node)
                                 seen.add(node)
                     elif typ == b'changesetexplicitdepth':
                         for key in (b'nodes', b'depth'):
                             if key not in spec:
                                 raise error.WireprotoCommandError(
                                     b'%s key not present in changesetexplicitdepth revision '
                                     b'specifier',
                                     (key,),
                                 )
                         for rev in repo.revs(
                             b'ancestors(%ln, %s)', spec[b'nodes'], spec[b'depth'] - 1
                         ):
                             node = cl.node(rev)
                             if node not in seen:
                                 nodes.append(node)
                                 seen.add(node)
                     elif typ == b'changesetdagrange':
                         for key in (b'roots', b'heads'):
                             if key not in spec:
                                 raise error.WireprotoCommandError(
                                     b'%s key not present in changesetdagrange revision '
                                     b'specifier',
                                     (key,),
                                 )
                         if not spec[b'heads']:
                             raise error.WireprotoCommandError(
                                 b'heads key in changesetdagrange cannot be empty'
                             )
                         if spec[b'roots']:
                             common = [n for n in spec[b'roots'] if clhasnode(n)]
                         else:
                             common = [nullid]
                         for n in discovery.outgoing(repo, common, spec[b'heads']).missing:
                             if n not in seen:
                                 nodes.append(n)
                                 seen.add(n)
                     else:
                         raise error.WireprotoCommandError(
                             b'unknown revision specifier type: %s', (typ,)
                         )
                 return nodes
             @wireprotocommand(b'branchmap', permission=b'pull')
             def branchmapv2(repo, proto):
                 yield {
                     encoding.fromlocal(k): v
                     for k, v in pycompat.iteritems(repo.branchmap())
                 }
             @wireprotocommand(b'capabilities', permission=b'pull')
             def capabilitiesv2(repo, proto):
                 yield _capabilitiesv2(repo, proto)
             @wireprotocommand(
                 b'changesetdata',
                 args={
                     b'revisions': {
                         b'type': b'list',
                         b'example': [
                             {
                                 b'type': b'changesetexplicit',
                                 b'nodes': [b'abcdef...'],
                             }
                         ],
                     },
                     b'fields': {
                         b'type': b'set',
                         b'default': set,
                         b'example': {b'parents', b'revision'},
                         b'validvalues': {b'bookmarks', b'parents', b'phase', b'revision'},
                     },
                 },
                 permission=b'pull',
             )
             def changesetdata(repo, proto, revisions, fields):
                 # TODO look for unknown fields and abort when they can't be serviced.
                 # This could probably be validated by dispatcher using validvalues.
                 cl = repo.changelog
                 outgoing = resolvenodes(repo, revisions)
                 publishing = repo.publishing()
                 if outgoing:
                     repo.hook(b'preoutgoing', throw=True, source=b'serve')
                 yield {
                     b'totalitems': len(outgoing),
                 }
                 # The phases of nodes already transferred to the client may have changed
                 # since the client last requested data. We send phase-only records
                 # for these revisions, if requested.
                 # TODO actually do this. We'll probably want to emit phase heads
                 # in the ancestry set of the outgoing revisions. This will ensure
                 # that phase updates within that set are seen.
                 if b'phase' in fields:
                     pass
                 nodebookmarks = {}
                 for mark, node in repo._bookmarks.items():
                     nodebookmarks.setdefault(node, set()).add(mark)
                 # It is already topologically sorted by revision number.
                 for node in outgoing:
                     d = {
                         b'node': node,
                     }
                     if b'parents' in fields:
                         d[b'parents'] = cl.parents(node)
                     if b'phase' in fields:
                         if publishing:
                             d[b'phase'] = b'public'
                         else:
                             ctx = repo[node]
                             d[b'phase'] = ctx.phasestr()
                     if b'bookmarks' in fields and node in nodebookmarks:
                         d[b'bookmarks'] = sorted(nodebookmarks[node])
                         del nodebookmarks[node]
                     followingmeta = []
                     followingdata = []
                     if b'revision' in fields:
                         revisiondata = cl.revision(node)
                         followingmeta.append((b'revision', len(revisiondata)))
                         followingdata.append(revisiondata)
                     # TODO make it possible for extensions to wrap a function or register
                     # a handler to service custom fields.
                     if followingmeta:
                         d[b'fieldsfollowing'] = followingmeta
                     yield d
                     for extra in followingdata:
                         yield extra
                 # If requested, send bookmarks from nodes that didn't have revision
                 # data sent so receiver is aware of any bookmark updates.
                 if b'bookmarks' in fields:
                     for node, marks in sorted(pycompat.iteritems(nodebookmarks)):
                         yield {
                             b'node': node,
                             b'bookmarks': sorted(marks),
                         }
             class FileAccessError(Exception):
                 """Represents an error accessing a specific file."""
                 def __init__(self, path, msg, args):
                     self.path = path
                     self.msg = msg
                     self.args = args
             def getfilestore(repo, proto, path):
                 """Obtain a file storage object for use with wire protocol.
                 Exists as a standalone function so extensions can monkeypatch to add
                 access control.
                 """
                 # This seems to work even if the file doesn't exist. So catch
                 # "empty" files and return an error.
                 fl = repo.file(path)
                 if not len(fl):
                     raise FileAccessError(path, b'unknown file: %s', (path,))
                 return fl
             def emitfilerevisions(repo, path, revisions, linknodes, fields):
                 for revision in revisions:
                     d = {
                         b'node': revision.node,
                     }
                     if b'parents' in fields:
                         d[b'parents'] = [revision.p1node, revision.p2node]
                     if b'linknode' in fields:
                         d[b'linknode'] = linknodes[revision.node]
                     followingmeta = []
                     followingdata = []
                     if b'revision' in fields:
                         if revision.revision is not None:
                             followingmeta.append((b'revision', len(revision.revision)))
                             followingdata.append(revision.revision)
                         else:
                             d[b'deltabasenode'] = revision.basenode
                             followingmeta.append((b'delta', len(revision.delta)))
                             followingdata.append(revision.delta)
                     if followingmeta:
                         d[b'fieldsfollowing'] = followingmeta
                     yield d
                     for extra in followingdata:
                         yield extra
             def makefilematcher(repo, pathfilter):
                 """Construct a matcher from a path filter dict."""
                 # Validate values.
                 if pathfilter:
                     for key in (b'include', b'exclude'):
                         for pattern in pathfilter.get(key, []):
                             if not pattern.startswith((b'path:', b'rootfilesin:')):
                                 raise error.WireprotoCommandError(
                                     b'%s pattern must begin with `path:` or `rootfilesin:`; '
                                     b'got %s',
                                     (key, pattern),
                                 )
                 if pathfilter:
                     matcher = matchmod.match(
                         repo.root,
                         b'',
                         include=pathfilter.get(b'include', []),
                         exclude=pathfilter.get(b'exclude', []),
                     )
                 else:
                     matcher = matchmod.match(repo.root, b'')
                 # Requested patterns could include files not in the local store. So
                 # filter those out.
                 return repo.narrowmatch(matcher)
             @wireprotocommand(
                 b'filedata',
                 args={
                     b'haveparents': {
                         b'type': b'bool',
                         b'default': lambda: False,
                         b'example': True,
                     },
                     b'nodes': {
                         b'type': b'list',
                         b'example': [b'0123456...'],
                     },
                     b'fields': {
                         b'type': b'set',
                         b'default': set,
                         b'example': {b'parents', b'revision'},
                         b'validvalues': {b'parents', b'revision', b'linknode'},
                     },
                     b'path': {
                         b'type': b'bytes',
                         b'example': b'foo.txt',
                     },
                 },
                 permission=b'pull',
                 # TODO censoring a file revision won't invalidate the cache.
                 # Figure out a way to take censoring into account when deriving
                 # the cache key.
                 cachekeyfn=makecommandcachekeyfn(b'filedata', 1, allargs=True),
             )
             def filedata(repo, proto, haveparents, nodes, fields, path):
                 # TODO this API allows access to file revisions that are attached to
                 # secret changesets. filesdata does not have this problem. Maybe this
                 # API should be deleted?
                 try:
                     # Extensions may wish to access the protocol handler.
                     store = getfilestore(repo, proto, path)
                 except FileAccessError as e:
                     raise error.WireprotoCommandError(e.msg, e.args)
                 clnode = repo.changelog.node
                 linknodes = {}
                 # Validate requested nodes.
                 for node in nodes:
                     try:
                         store.rev(node)
                     except error.LookupError:
                         raise error.WireprotoCommandError(
                             b'unknown file node: %s', (hex(node),)
                         )
                     # TODO by creating the filectx against a specific file revision
                     # instead of changeset, linkrev() is always used. This is wrong for
                     # cases where linkrev() may refer to a hidden changeset. But since this
                     # API doesn't know anything about changesets, we're not sure how to
                     # disambiguate the linknode. Perhaps we should delete this API?
                     fctx = repo.filectx(path, fileid=node)
                     linknodes[node] = clnode(fctx.introrev())
                 revisions = store.emitrevisions(
                     nodes,
                     revisiondata=b'revision' in fields,
                     assumehaveparentrevisions=haveparents,
                 )
                 yield {
                     b'totalitems': len(nodes),
                 }
                 for o in emitfilerevisions(repo, path, revisions, linknodes, fields):
                     yield o
             def filesdatacapabilities(repo, proto):
                 batchsize = repo.ui.configint(
                     b'experimental', b'server.filesdata.recommended-batch-size'
                 )
                 return {
                     b'recommendedbatchsize': batchsize,
                 }
             @wireprotocommand(
                 b'filesdata',
                 args={
                     b'haveparents': {
                         b'type': b'bool',
                         b'default': lambda: False,
                         b'example': True,
                     },
                     b'fields': {
                         b'type': b'set',
                         b'default': set,
                         b'example': {b'parents', b'revision'},
                         b'validvalues': {
                             b'firstchangeset',
                             b'linknode',
                             b'parents',
                             b'revision',
                         },
                     },
                     b'pathfilter': {
                         b'type': b'dict',
                         b'default': lambda: None,
                         b'example': {b'include': [b'path:tests']},
                     },
                     b'revisions': {
                         b'type': b'list',
                         b'example': [
                             {
                                 b'type': b'changesetexplicit',
                                 b'nodes': [b'abcdef...'],
                             }
                         ],
                     },
                 },
                 permission=b'pull',
                 # TODO censoring a file revision won't invalidate the cache.
                 # Figure out a way to take censoring into account when deriving
                 # the cache key.
                 cachekeyfn=makecommandcachekeyfn(b'filesdata', 1, allargs=True),
                 extracapabilitiesfn=filesdatacapabilities,
             )
             def filesdata(repo, proto, haveparents, fields, pathfilter, revisions):
                 # TODO This should operate on a repo that exposes obsolete changesets. There
                 # is a race between a client making a push that obsoletes a changeset and
                 # another client fetching files data for that changeset. If a client has a
                 # changeset, it should probably be allowed to access files data for that
                 # changeset.
                 outgoing = resolvenodes(repo, revisions)
                 filematcher = makefilematcher(repo, pathfilter)
                 # path -> {fnode: linknode}
                 fnodes = collections.defaultdict(dict)
                 # We collect the set of relevant file revisions by iterating the changeset
                 # revisions and either walking the set of files recorded in the changeset
                 # or by walking the manifest at that revision. There is probably room for a
                 # storage-level API to request this data, as it can be expensive to compute
                 # and would benefit from caching or alternate storage from what revlogs
                 # provide.
                 for node in outgoing:
                     ctx = repo[node]
                     mctx = ctx.manifestctx()
                     md = mctx.read()
                     if haveparents:
                         checkpaths = ctx.files()
                     else:
                         checkpaths = md.keys()
                     for path in checkpaths:
                         fnode = md[path]
                         if path in fnodes and fnode in fnodes[path]:
                             continue
                         if not filematcher(path):
                             continue
                         fnodes[path].setdefault(fnode, node)
                 yield {
                     b'totalpaths': len(fnodes),
                     b'totalitems': sum(len(v) for v in fnodes.values()),
                 }
                 for path, filenodes in sorted(fnodes.items()):
                     try:
                         store = getfilestore(repo, proto, path)
                     except FileAccessError as e:
                         raise error.WireprotoCommandError(e.msg, e.args)
                     yield {
                         b'path': path,
                         b'totalitems': len(filenodes),
                     }
                     revisions = store.emitrevisions(
                         filenodes.keys(),
                         revisiondata=b'revision' in fields,
                         assumehaveparentrevisions=haveparents,
                     )
                     for o in emitfilerevisions(repo, path, revisions, filenodes, fields):
                         yield o
             @wireprotocommand(
                 b'heads',
                 args={
                     b'publiconly': {
                         b'type': b'bool',
                         b'default': lambda: False,
                         b'example': False,
                     },
                 },
                 permission=b'pull',
             )
             def headsv2(repo, proto, publiconly):
                 if publiconly:
                     repo = repo.filtered(b'immutable')
                 yield repo.heads()
             @wireprotocommand(
                 b'known',
                 args={
                     b'nodes': {
                         b'type': b'list',
                         b'default': list,
                         b'example': [b'deadbeef'],
                     },
                 },
                 permission=b'pull',
             )
             def knownv2(repo, proto, nodes):
                 result = b''.join(b'1' if n else b'0' for n in repo.known(nodes))
                 yield result
             @wireprotocommand(
                 b'listkeys',
                 args={
                     b'namespace': {
                         b'type': b'bytes',
                         b'example': b'ns',
                     },
                 },
                 permission=b'pull',
             )
             def listkeysv2(repo, proto, namespace):
                 keys = repo.listkeys(encoding.tolocal(namespace))
                 keys = {
                     encoding.fromlocal(k): encoding.fromlocal(v)
                     for k, v in pycompat.iteritems(keys)
                 }
                 yield keys
             @wireprotocommand(
                 b'lookup',
                 args={
                     b'key': {
                         b'type': b'bytes',
                         b'example': b'foo',
                     },
                 },
                 permission=b'pull',
             )
             def lookupv2(repo, proto, key):
                 key = encoding.tolocal(key)
                 # TODO handle exception.
                 node = repo.lookup(key)
                 yield node
             def manifestdatacapabilities(repo, proto):
                 batchsize = repo.ui.configint(
                     b'experimental', b'server.manifestdata.recommended-batch-size'
                 )
                 return {
                     b'recommendedbatchsize': batchsize,
                 }
             @wireprotocommand(
                 b'manifestdata',
                 args={
                     b'nodes': {
                         b'type': b'list',
                         b'example': [b'0123456...'],
                     },
                     b'haveparents': {
                         b'type': b'bool',
                         b'default': lambda: False,
                         b'example': True,
                     },
                     b'fields': {
                         b'type': b'set',
                         b'default': set,
                         b'example': {b'parents', b'revision'},
                         b'validvalues': {b'parents', b'revision'},
                     },
                     b'tree': {
                         b'type': b'bytes',
                         b'example': b'',
                     },
                 },
                 permission=b'pull',
                 cachekeyfn=makecommandcachekeyfn(b'manifestdata', 1, allargs=True),
                 extracapabilitiesfn=manifestdatacapabilities,
             )
             def manifestdata(repo, proto, haveparents, nodes, fields, tree):
                 store = repo.manifestlog.getstorage(tree)
                 # Validate the node is known and abort on unknown revisions.
                 for node in nodes:
                     try:
                         store.rev(node)
                     except error.LookupError:
                         raise error.WireprotoCommandError(b'unknown node: %s', (node,))
                 revisions = store.emitrevisions(
                     nodes,
                     revisiondata=b'revision' in fields,
                     assumehaveparentrevisions=haveparents,
                 )
                 yield {
                     b'totalitems': len(nodes),
                 }
                 for revision in revisions:
                     d = {
                         b'node': revision.node,
                     }
                     if b'parents' in fields:
                         d[b'parents'] = [revision.p1node, revision.p2node]
                     followingmeta = []
                     followingdata = []
                     if b'revision' in fields:
                         if revision.revision is not None:
                             followingmeta.append((b'revision', len(revision.revision)))
                             followingdata.append(revision.revision)
                         else:
                             d[b'deltabasenode'] = revision.basenode
                             followingmeta.append((b'delta', len(revision.delta)))
                             followingdata.append(revision.delta)
                     if followingmeta:
                         d[b'fieldsfollowing'] = followingmeta
                     yield d
                     for extra in followingdata:
                         yield extra
             @wireprotocommand(
                 b'pushkey',
                 args={
                     b'namespace': {
                         b'type': b'bytes',
                         b'example': b'ns',
                     },
                     b'key': {
                         b'type': b'bytes',
                         b'example': b'key',
                     },
                     b'old': {
                         b'type': b'bytes',
                         b'example': b'old',
                     },
                     b'new': {
                         b'type': b'bytes',
                         b'example': b'new',
                     },
                 },
                 permission=b'push',
             )
             def pushkeyv2(repo, proto, namespace, key, old, new):
                 # TODO handle ui output redirection
                 yield repo.pushkey(
                     encoding.tolocal(namespace),
                     encoding.tolocal(key),
                     encoding.tolocal(old),
                     encoding.tolocal(new),
                 )
             @wireprotocommand(
                 b'rawstorefiledata',
                 args={
                     b'files': {
                         b'type': b'list',
                         b'example': [b'changelog', b'manifestlog'],
                     },
                     b'pathfilter': {
                         b'type': b'list',
                         b'default': lambda: None,
                         b'example': {b'include': [b'path:tests']},
                     },
                 },
                 permission=b'pull',
             )
             def rawstorefiledata(repo, proto, files, pathfilter):
                 if not streamclone.allowservergeneration(repo):
                     raise error.WireprotoCommandError(b'stream clone is disabled')
                 # TODO support dynamically advertising what store files "sets" are
                 # available. For now, we support changelog, manifestlog, and files.
                 files = set(files)
                 allowedfiles = {b'changelog', b'manifestlog'}
                 unsupported = files - allowedfiles
                 if unsupported:
                     raise error.WireprotoCommandError(
                         b'unknown file type: %s', (b', '.join(sorted(unsupported)),)
                     )
                 with repo.lock():
                     topfiles = list(repo.store.topfiles())
                 sendfiles = []
                 totalsize = 0
                 # TODO this is a bunch of storage layer interface abstractions because
                 # it assumes revlogs.
-                for name, encodedname, size in topfiles:
+                for rl_type, name, encodedname, size in topfiles:
+                    # XXX use the `rl_type` for that
                     if b'changelog' in files and name.startswith(b'00changelog'):
                         pass
                     elif b'manifestlog' in files and name.startswith(b'00manifest'):
                         pass
                     else:
                         continue
                     sendfiles.append((b'store', name, size))
                     totalsize += size
                 yield {
                     b'filecount': len(sendfiles),
                     b'totalsize': totalsize,
                 }
                 for location, name, size in sendfiles:
                     yield {
                         b'location': location,
                         b'path': name,
                         b'size': size,
                     }
                     # We have to use a closure for this to ensure the context manager is
                     # closed only after sending the final chunk.
                     def getfiledata():
                         with repo.svfs(name, b'rb', auditpath=False) as fh:
                             for chunk in util.filechunkiter(fh, limit=size):
                                 yield chunk
                     yield wireprototypes.indefinitebytestringresponse(getfiledata())

tests/test-persistent-nodemap.t

0 +2 -2

             ===================================
             Test the persistent on-disk nodemap
             ===================================
             #if no-rust
               $ cat << EOF >> $HGRCPATH
               > [format]
               > use-persistent-nodemap=yes
               > [devel]
               > persistent-nodemap=yes
               > EOF
             #endif
               $ hg init test-repo --config storage.revlog.persistent-nodemap.slow-path=allow
               $ cd test-repo
             Check handling of the default slow-path value
             #if no-pure no-rust
               $ hg id
               abort: accessing `persistent-nodemap` repository without associated fast implementation.
               (check `hg help config.format.use-persistent-nodemap` for details)
               [255]
             Unlock further check (we are here to test the feature)
               $ cat << EOF >> $HGRCPATH
               > [storage]
               > # to avoid spamming the test
               > revlog.persistent-nodemap.slow-path=allow
               > EOF
             #endif
             #if rust
             Regression test for a previous bug in Rust/C FFI for the `Revlog_CAPI` capsule:
             in places where `mercurial/cext/revlog.c` function signatures use `Py_ssize_t`
             (64 bits on Linux x86_64), corresponding declarations in `rust/hg-cpython/src/cindex.rs`
             incorrectly used `libc::c_int` (32 bits).
             As a result, -1 passed from Rust for the null revision became 4294967295 in C.
               $ hg log -r 00000000
               changeset:   -1:000000000000
               tag:         tip
               user:
               date:        Thu Jan 01 00:00:00 1970 +0000
             #endif
               $ hg debugformat
               format-variant     repo
               fncache:            yes
               dotencode:          yes
               generaldelta:       yes
               share-safe:          no
               sparserevlog:       yes
               persistent-nodemap: yes
               copies-sdc:          no
               revlog-v2:           no
               plain-cl-delta:     yes
               compression:        zlib (no-zstd !)
               compression:        zstd (zstd !)
               compression-level:  default
               $ hg debugbuilddag .+5000 --new-file
               $ hg debugnodemap --metadata
               uid: ???????????????? (glob)
               tip-rev: 5000
               tip-node: 6b02b8c7b96654c25e86ba69eda198d7e6ad8b3c
               data-length: 121088
               data-unused: 0
               data-unused: 0.000%
               $ f --size .hg/store/00changelog.n
               .hg/store/00changelog.n: size=70
             Simple lookup works
               $ ANYNODE=`hg log --template '{node|short}\n' --rev tip`
               $ hg log -r "$ANYNODE" --template '{rev}\n'
             #if rust
               $ f --sha256 .hg/store/00changelog-*.nd
               .hg/store/00changelog-????????????????.nd: sha256=2e029d3200bd1a986b32784fc2ef1a3bd60dc331f025718bcf5ff44d93f026fd (glob)
               $ f --sha256 .hg/store/00manifest-*.nd
               .hg/store/00manifest-????????????????.nd: sha256=97117b1c064ea2f86664a124589e47db0e254e8d34739b5c5cc5bf31c9da2b51 (glob)
               $ hg debugnodemap --dump-new | f --sha256 --size
               size=121088, sha256=2e029d3200bd1a986b32784fc2ef1a3bd60dc331f025718bcf5ff44d93f026fd
               $ hg debugnodemap --dump-disk | f --sha256 --bytes=256 --hexdump --size
               size=121088, sha256=2e029d3200bd1a986b32784fc2ef1a3bd60dc331f025718bcf5ff44d93f026fd
 : 00 00 00 91 00 00 00 20 00 00 00 bb 00 00 00 e7 |....... ........|
 : 00 00 00 66 00 00 00 a1 00 00 01 13 00 00 01 22 |...f..........."|
 : 00 00 00 23 00 00 00 fc 00 00 00 ba 00 00 00 5e |...#...........^|
 : 00 00 00 df 00 00 01 4e 00 00 01 65 00 00 00 ab |.......N...e....|
 : 00 00 00 a9 00 00 00 95 00 00 00 73 00 00 00 38 |...........s...8|
 : 00 00 00 cc 00 00 00 92 00 00 00 90 00 00 00 69 |...............i|
 : 00 00 00 ec 00 00 00 8d 00 00 01 4f 00 00 00 12 |...........O....|
 : 00 00 02 0c 00 00 00 77 00 00 00 9c 00 00 00 8f |.......w........|
 : 00 00 00 d5 00 00 00 6b 00 00 00 48 00 00 00 b3 |.......k...H....|
 : 00 00 00 e5 00 00 00 b5 00 00 00 8e 00 00 00 ad |................|
 a0: 00 00 00 7b 00 00 00 7c 00 00 00 0b 00 00 00 2b |...{...|.......+|
 b0: 00 00 00 c6 00 00 00 1e 00 00 01 08 00 00 00 11 |................|
 c0: 00 00 01 30 00 00 00 26 00 00 01 9c 00 00 00 35 |...0...&.......5|
 d0: 00 00 00 b8 00 00 01 31 00 00 00 2c 00 00 00 55 |.......1...,...U|
 e0: 00 00 00 8a 00 00 00 9a 00 00 00 0c 00 00 01 1e |................|
 f0: 00 00 00 a4 00 00 00 83 00 00 00 c9 00 00 00 8c |................|
             #else
               $ f --sha256 .hg/store/00changelog-*.nd
               .hg/store/00changelog-????????????????.nd: sha256=f544f5462ff46097432caf6d764091f6d8c46d6121be315ead8576d548c9dd79 (glob)
               $ hg debugnodemap --dump-new | f --sha256 --size
               size=121088, sha256=f544f5462ff46097432caf6d764091f6d8c46d6121be315ead8576d548c9dd79
               $ hg debugnodemap --dump-disk | f --sha256 --bytes=256 --hexdump --size
               size=121088, sha256=f544f5462ff46097432caf6d764091f6d8c46d6121be315ead8576d548c9dd79
 : ff ff ff ff ff ff ff ff ff ff ff ff ff ff ff ff |................|
 : ff ff ff ff ff ff ff ff ff ff ff ff ff ff ff ff |................|
 : ff ff ff ff ff ff f5 06 ff ff ff ff ff ff f3 e7 |................|
 : ff ff ef ca ff ff ff ff ff ff ff ff ff ff ff ff |................|
 : ff ff ff ff ff ff ff ff ff ff ff ff ff ff ff ff |................|
 : ff ff ff ff ff ff ff ff ff ff ff ff ff ff ed 08 |................|
 : ff ff ed 66 ff ff ff ff ff ff ff ff ff ff ff ff |...f............|
 : ff ff ff ff ff ff ff ff ff ff ff ff ff ff ff ff |................|
 : ff ff ff ff ff ff ff ff ff ff ff ff ff ff ff ff |................|
 : ff ff ff ff ff ff ff ff ff ff ff ff ff ff f6 ed |................|
 a0: ff ff ff ff ff ff fe 61 ff ff ff ff ff ff ff ff |.......a........|
 b0: ff ff ff ff ff ff ff ff ff ff ff ff ff ff ff ff |................|
 c0: ff ff ff ff ff ff ff ff ff ff ff ff ff ff ff ff |................|
 d0: ff ff ff ff ff ff ff ff ff ff ff ff ff ff ff ff |................|
 e0: ff ff ff ff ff ff ff ff ff ff ff ff ff ff f1 02 |................|
 f0: ff ff ff ff ff ff ed 1b ff ff ff ff ff ff ff ff |................|
             #endif
               $ hg debugnodemap --check
               revision in index:   5001
               revision in nodemap: 5001
             add a new commit
               $ hg up
 files updated, 0 files merged, 0 files removed, 0 files unresolved
               $ echo foo > foo
               $ hg add foo
             Check slow-path config value handling
             -------------------------------------
             #if no-pure no-rust
               $ hg id --config "storage.revlog.persistent-nodemap.slow-path=invalid-value"
               unknown value for config "storage.revlog.persistent-nodemap.slow-path": "invalid-value"
               falling back to default value: abort
               abort: accessing `persistent-nodemap` repository without associated fast implementation.
               (check `hg help config.format.use-persistent-nodemap` for details)
               [255]
               $ hg log -r . --config "storage.revlog.persistent-nodemap.slow-path=warn"
               warning: accessing `persistent-nodemap` repository without associated fast implementation.
               (check `hg help config.format.use-persistent-nodemap` for details)
               changeset:   5000:6b02b8c7b966
               tag:         tip
               user:        debugbuilddag
               date:        Thu Jan 01 01:23:20 1970 +0000
               summary:     r5000
               $ hg ci -m 'foo' --config "storage.revlog.persistent-nodemap.slow-path=abort"
               abort: accessing `persistent-nodemap` repository without associated fast implementation.
               (check `hg help config.format.use-persistent-nodemap` for details)
               [255]
             #else
               $ hg id --config "storage.revlog.persistent-nodemap.slow-path=invalid-value"
               unknown value for config "storage.revlog.persistent-nodemap.slow-path": "invalid-value"
               falling back to default value: abort
 b02b8c7b966+ tip
             #endif
               $ hg ci -m 'foo'
             #if no-pure no-rust
               $ hg debugnodemap --metadata
               uid: ???????????????? (glob)
               tip-rev: 5001
               tip-node: 16395c3cf7e231394735e6b1717823ada303fb0c
               data-length: 121088
               data-unused: 0
               data-unused: 0.000%
             #else
               $ hg debugnodemap --metadata
               uid: ???????????????? (glob)
               tip-rev: 5001
               tip-node: 16395c3cf7e231394735e6b1717823ada303fb0c
               data-length: 121344
               data-unused: 256
               data-unused: 0.211%
             #endif
               $ f --size .hg/store/00changelog.n
               .hg/store/00changelog.n: size=70
             (The pure code use the debug code that perform incremental update, the C code reencode from scratch)
             #if pure
               $ f --sha256 .hg/store/00changelog-*.nd --size
               .hg/store/00changelog-????????????????.nd: size=121344, sha256=cce54c5da5bde3ad72a4938673ed4064c86231b9c64376b082b163fdb20f8f66 (glob)
             #endif
             #if rust
               $ f --sha256 .hg/store/00changelog-*.nd --size
               .hg/store/00changelog-????????????????.nd: size=121344, sha256=952b042fcf614ceb37b542b1b723e04f18f83efe99bee4e0f5ccd232ef470e58 (glob)
             #endif
             #if no-pure no-rust
               $ f --sha256 .hg/store/00changelog-*.nd --size
               .hg/store/00changelog-????????????????.nd: size=121088, sha256=df7c06a035b96cb28c7287d349d603baef43240be7736fe34eea419a49702e17 (glob)
             #endif
               $ hg debugnodemap --check
               revision in index:   5002
               revision in nodemap: 5002
             Test code path without mmap
             ---------------------------
               $ echo bar > bar
               $ hg add bar
               $ hg ci -m 'bar' --config storage.revlog.persistent-nodemap.mmap=no
               $ hg debugnodemap --check --config storage.revlog.persistent-nodemap.mmap=yes
               revision in index:   5003
               revision in nodemap: 5003
               $ hg debugnodemap --check --config storage.revlog.persistent-nodemap.mmap=no
               revision in index:   5003
               revision in nodemap: 5003
             #if pure
               $ hg debugnodemap --metadata
               uid: ???????????????? (glob)
               tip-rev: 5002
               tip-node: 880b18d239dfa9f632413a2071bfdbcc4806a4fd
               data-length: 121600
               data-unused: 512
               data-unused: 0.421%
               $ f --sha256 .hg/store/00changelog-*.nd --size
               .hg/store/00changelog-????????????????.nd: size=121600, sha256=def52503d049ccb823974af313a98a935319ba61f40f3aa06a8be4d35c215054 (glob)
             #endif
             #if rust
               $ hg debugnodemap --metadata
               uid: ???????????????? (glob)
               tip-rev: 5002
               tip-node: 880b18d239dfa9f632413a2071bfdbcc4806a4fd
               data-length: 121600
               data-unused: 512
               data-unused: 0.421%
               $ f --sha256 .hg/store/00changelog-*.nd --size
               .hg/store/00changelog-????????????????.nd: size=121600, sha256=dacf5b5f1d4585fee7527d0e67cad5b1ba0930e6a0928f650f779aefb04ce3fb (glob)
             #endif
             #if no-pure no-rust
               $ hg debugnodemap --metadata
               uid: ???????????????? (glob)
               tip-rev: 5002
               tip-node: 880b18d239dfa9f632413a2071bfdbcc4806a4fd
               data-length: 121088
               data-unused: 0
               data-unused: 0.000%
               $ f --sha256 .hg/store/00changelog-*.nd --size
               .hg/store/00changelog-????????????????.nd: size=121088, sha256=59fcede3e3cc587755916ceed29e3c33748cd1aa7d2f91828ac83e7979d935e8 (glob)
             #endif
             Test force warming the cache
               $ rm .hg/store/00changelog.n
               $ hg debugnodemap --metadata
               $ hg debugupdatecache
             #if pure
               $ hg debugnodemap --metadata
               uid: ???????????????? (glob)
               tip-rev: 5002
               tip-node: 880b18d239dfa9f632413a2071bfdbcc4806a4fd
               data-length: 121088
               data-unused: 0
               data-unused: 0.000%
             #else
               $ hg debugnodemap --metadata
               uid: ???????????????? (glob)
               tip-rev: 5002
               tip-node: 880b18d239dfa9f632413a2071bfdbcc4806a4fd
               data-length: 121088
               data-unused: 0
               data-unused: 0.000%
             #endif
             Check out of sync nodemap
             =========================
             First copy old data on the side.
               $ mkdir ../tmp-copies
               $ cp .hg/store/00changelog-????????????????.nd .hg/store/00changelog.n ../tmp-copies
             Nodemap lagging behind
             ----------------------
             make a new commit
               $ echo bar2 > bar
               $ hg ci -m 'bar2'
               $ NODE=`hg log -r tip -T '{node}\n'`
               $ hg log -r "$NODE" -T '{rev}\n'
             If the nodemap is lagging behind, it can catch up fine
               $ hg debugnodemap --metadata
               uid: ???????????????? (glob)
               tip-rev: 5003
               tip-node: c9329770f979ade2d16912267c38ba5f82fd37b3
               data-length: 121344 (pure !)
               data-length: 121344 (rust !)
               data-length: 121152 (no-rust no-pure !)
               data-unused: 192 (pure !)
               data-unused: 192 (rust !)
               data-unused: 0 (no-rust no-pure !)
               data-unused: 0.158% (pure !)
               data-unused: 0.158% (rust !)
               data-unused: 0.000% (no-rust no-pure !)
               $ cp -f ../tmp-copies/* .hg/store/
               $ hg debugnodemap --metadata
               uid: ???????????????? (glob)
               tip-rev: 5002
               tip-node: 880b18d239dfa9f632413a2071bfdbcc4806a4fd
               data-length: 121088
               data-unused: 0
               data-unused: 0.000%
               $ hg log -r "$NODE" -T '{rev}\n'
             changelog altered
             -----------------
             If the nodemap is not gated behind a requirements, an unaware client can alter
             the repository so the revlog used to generate the nodemap is not longer
             compatible with the persistent nodemap. We need to detect that.
               $ hg up "$NODE~5"
 files updated, 0 files merged, 4 files removed, 0 files unresolved
               $ echo bar > babar
               $ hg add babar
               $ hg ci -m 'babar'
               created new head
               $ OTHERNODE=`hg log -r tip -T '{node}\n'`
               $ hg log -r "$OTHERNODE" -T '{rev}\n'
               $ hg --config extensions.strip= strip --rev "$NODE~1" --no-backup
             the nodemap should detect the changelog have been tampered with and recover.
               $ hg debugnodemap --metadata
               uid: ???????????????? (glob)
               tip-rev: 5002
               tip-node: b355ef8adce0949b8bdf6afc72ca853740d65944
               data-length: 121536 (pure !)
               data-length: 121088 (rust !)
               data-length: 121088 (no-pure no-rust !)
               data-unused: 448 (pure !)
               data-unused: 0 (rust !)
               data-unused: 0 (no-pure no-rust !)
               data-unused: 0.000% (rust !)
               data-unused: 0.369% (pure !)
               data-unused: 0.000% (no-pure no-rust !)
               $ cp -f ../tmp-copies/* .hg/store/
               $ hg debugnodemap --metadata
               uid: ???????????????? (glob)
               tip-rev: 5002
               tip-node: 880b18d239dfa9f632413a2071bfdbcc4806a4fd
               data-length: 121088
               data-unused: 0
               data-unused: 0.000%
               $ hg log -r "$OTHERNODE" -T '{rev}\n'
             missing data file
             -----------------
               $ UUID=`hg debugnodemap --metadata| grep 'uid:' | \
               > sed 's/uid: //'`
               $ FILE=.hg/store/00changelog-"${UUID}".nd
               $ mv $FILE ../tmp-data-file
               $ cp .hg/store/00changelog.n ../tmp-docket
             mercurial don't crash
               $ hg log -r .
               changeset:   5002:b355ef8adce0
               tag:         tip
               parent:      4998:d918ad6d18d3
               user:        test
               date:        Thu Jan 01 00:00:00 1970 +0000
               summary:     babar
               $ hg debugnodemap --metadata
               $ hg debugupdatecache
               $ hg debugnodemap --metadata
               uid: * (glob)
               tip-rev: 5002
               tip-node: b355ef8adce0949b8bdf6afc72ca853740d65944
               data-length: 121088
               data-unused: 0
               data-unused: 0.000%
               $ mv ../tmp-data-file $FILE
               $ mv ../tmp-docket .hg/store/00changelog.n
             Check transaction related property
             ==================================
             An up to date nodemap should be available to shell hooks,
               $ echo dsljfl > a
               $ hg add a
               $ hg ci -m a
               $ hg debugnodemap --metadata
               uid: ???????????????? (glob)
               tip-rev: 5003
               tip-node: a52c5079765b5865d97b993b303a18740113bbb2
               data-length: 121088
               data-unused: 0
               data-unused: 0.000%
               $ echo babar2 > babar
               $ hg ci -m 'babar2' --config "hooks.pretxnclose.nodemap-test=hg debugnodemap --metadata"
               uid: ???????????????? (glob)
               tip-rev: 5004
               tip-node: 2f5fb1c06a16834c5679d672e90da7c5f3b1a984
               data-length: 121280 (pure !)
               data-length: 121280 (rust !)
               data-length: 121088 (no-pure no-rust !)
               data-unused: 192 (pure !)
               data-unused: 192 (rust !)
               data-unused: 0 (no-pure no-rust !)
               data-unused: 0.158% (pure !)
               data-unused: 0.158% (rust !)
               data-unused: 0.000% (no-pure no-rust !)
               $ hg debugnodemap --metadata
               uid: ???????????????? (glob)
               tip-rev: 5004
               tip-node: 2f5fb1c06a16834c5679d672e90da7c5f3b1a984
               data-length: 121280 (pure !)
               data-length: 121280 (rust !)
               data-length: 121088 (no-pure no-rust !)
               data-unused: 192 (pure !)
               data-unused: 192 (rust !)
               data-unused: 0 (no-pure no-rust !)
               data-unused: 0.158% (pure !)
               data-unused: 0.158% (rust !)
               data-unused: 0.000% (no-pure no-rust !)
             Another process does not see the pending nodemap content during run.
               $ PATH=$RUNTESTDIR/testlib/:$PATH
               $ echo qpoasp > a
               $ hg ci -m a2 \
               > --config "hooks.pretxnclose=wait-on-file 20 sync-repo-read sync-txn-pending" \
               > --config "hooks.txnclose=touch sync-txn-close" > output.txt 2>&1 &
             (read the repository while the commit transaction is pending)
               $ wait-on-file 20 sync-txn-pending && \
               > hg debugnodemap --metadata && \
               > wait-on-file 20 sync-txn-close sync-repo-read
               uid: ???????????????? (glob)
               tip-rev: 5004
               tip-node: 2f5fb1c06a16834c5679d672e90da7c5f3b1a984
               data-length: 121280 (pure !)
               data-length: 121280 (rust !)
               data-length: 121088 (no-pure no-rust !)
               data-unused: 192 (pure !)
               data-unused: 192 (rust !)
               data-unused: 0 (no-pure no-rust !)
               data-unused: 0.158% (pure !)
               data-unused: 0.158% (rust !)
               data-unused: 0.000% (no-pure no-rust !)
               $ hg debugnodemap --metadata
               uid: ???????????????? (glob)
               tip-rev: 5005
               tip-node: 90d5d3ba2fc47db50f712570487cb261a68c8ffe
               data-length: 121536 (pure !)
               data-length: 121536 (rust !)
               data-length: 121088 (no-pure no-rust !)
               data-unused: 448 (pure !)
               data-unused: 448 (rust !)
               data-unused: 0 (no-pure no-rust !)
               data-unused: 0.369% (pure !)
               data-unused: 0.369% (rust !)
               data-unused: 0.000% (no-pure no-rust !)
               $ cat output.txt
             Check that a failing transaction will properly revert the data
               $ echo plakfe > a
               $ f --size --sha256 .hg/store/00changelog-*.nd
               .hg/store/00changelog-????????????????.nd: size=121536, sha256=bb414468d225cf52d69132e1237afba34d4346ee2eb81b505027e6197b107f03 (glob) (pure !)
               .hg/store/00changelog-????????????????.nd: size=121536, sha256=909ac727bc4d1c0fda5f7bff3c620c98bd4a2967c143405a1503439e33b377da (glob) (rust !)
               .hg/store/00changelog-????????????????.nd: size=121088, sha256=342d36d30d86dde67d3cb6c002606c4a75bcad665595d941493845066d9c8ee0 (glob) (no-pure no-rust !)
               $ hg ci -m a3 --config "extensions.abort=$RUNTESTDIR/testlib/crash_transaction_late.py"
               transaction abort!
               rollback completed
               abort: This is a late abort
               [255]
               $ hg debugnodemap --metadata
               uid: ???????????????? (glob)
               tip-rev: 5005
               tip-node: 90d5d3ba2fc47db50f712570487cb261a68c8ffe
               data-length: 121536 (pure !)
               data-length: 121536 (rust !)
               data-length: 121088 (no-pure no-rust !)
               data-unused: 448 (pure !)
               data-unused: 448 (rust !)
               data-unused: 0 (no-pure no-rust !)
               data-unused: 0.369% (pure !)
               data-unused: 0.369% (rust !)
               data-unused: 0.000% (no-pure no-rust !)
               $ f --size --sha256 .hg/store/00changelog-*.nd
               .hg/store/00changelog-????????????????.nd: size=121536, sha256=bb414468d225cf52d69132e1237afba34d4346ee2eb81b505027e6197b107f03 (glob) (pure !)
               .hg/store/00changelog-????????????????.nd: size=121536, sha256=909ac727bc4d1c0fda5f7bff3c620c98bd4a2967c143405a1503439e33b377da (glob) (rust !)
               .hg/store/00changelog-????????????????.nd: size=121088, sha256=342d36d30d86dde67d3cb6c002606c4a75bcad665595d941493845066d9c8ee0 (glob) (no-pure no-rust !)
             Check that removing content does not confuse the nodemap
             --------------------------------------------------------
             removing data with rollback
               $ echo aso > a
               $ hg ci -m a4
               $ hg rollback
               repository tip rolled back to revision 5005 (undo commit)
               working directory now based on revision 5005
               $ hg id -r .
 d5d3ba2fc4 tip
             roming data with strip
               $ echo aso > a
               $ hg ci -m a4
               $ hg --config extensions.strip= strip -r . --no-backup
 files updated, 0 files merged, 0 files removed, 0 files unresolved
               $ hg id -r . --traceback
 d5d3ba2fc4 tip
             Test upgrade / downgrade
             ========================
             downgrading
               $ cat << EOF >> .hg/hgrc
               > [format]
               > use-persistent-nodemap=no
               > EOF
               $ hg debugformat -v
               format-variant     repo config default
               fncache:            yes    yes     yes
               dotencode:          yes    yes     yes
               generaldelta:       yes    yes     yes
               share-safe:          no     no      no
               sparserevlog:       yes    yes     yes
               persistent-nodemap: yes     no      no
               copies-sdc:          no     no      no
               revlog-v2:           no     no      no
               plain-cl-delta:     yes    yes     yes
               compression:        zlib   zlib    zlib (no-zstd !)
               compression:        zstd   zstd    zstd (zstd !)
               compression-level:  default default default
               $ hg debugupgraderepo --run --no-backup
               upgrade will perform the following actions:
               requirements
                  preserved: dotencode, fncache, generaldelta, revlogv1, sparserevlog, store (no-zstd !)
                  preserved: dotencode, fncache, generaldelta, revlog-compression-zstd, revlogv1, sparserevlog, store (zstd !)
                  removed: persistent-nodemap
               processed revlogs:
                 - all-filelogs
                 - changelog
                 - manifest
               beginning upgrade...
               repository locked and read-only
               creating temporary repository to stage upgraded data: $TESTTMP/test-repo/.hg/upgrade.* (glob)
               (it is safe to interrupt this process any time before data migration completes)
               downgrading repository to not use persistent nodemap feature
               removing temporary repository $TESTTMP/test-repo/.hg/upgrade.* (glob)
               $ ls -1 .hg/store/ | egrep '00(changelog|manifest)(\.n|-.*\.nd)'
 changelog-*.nd (glob)
 manifest-*.nd (glob)
               undo.backup.00changelog.n
               undo.backup.00manifest.n
               $ hg debugnodemap --metadata
             upgrading
               $ cat << EOF >> .hg/hgrc
               > [format]
               > use-persistent-nodemap=yes
               > EOF
               $ hg debugformat -v
               format-variant     repo config default
               fncache:            yes    yes     yes
               dotencode:          yes    yes     yes
               generaldelta:       yes    yes     yes
               share-safe:          no     no      no
               sparserevlog:       yes    yes     yes
               persistent-nodemap:  no    yes      no
               copies-sdc:          no     no      no
               revlog-v2:           no     no      no
               plain-cl-delta:     yes    yes     yes
               compression:        zlib   zlib    zlib (no-zstd !)
               compression:        zstd   zstd    zstd (zstd !)
               compression-level:  default default default
               $ hg debugupgraderepo --run --no-backup
               upgrade will perform the following actions:
               requirements
                  preserved: dotencode, fncache, generaldelta, revlogv1, sparserevlog, store (no-zstd !)
                  preserved: dotencode, fncache, generaldelta, revlog-compression-zstd, revlogv1, sparserevlog, store (zstd !)
                  added: persistent-nodemap
               persistent-nodemap
                  Speedup revision lookup by node id.
               processed revlogs:
                 - all-filelogs
                 - changelog
                 - manifest
               beginning upgrade...
               repository locked and read-only
               creating temporary repository to stage upgraded data: $TESTTMP/test-repo/.hg/upgrade.* (glob)
               (it is safe to interrupt this process any time before data migration completes)
               upgrading repository to use persistent nodemap feature
               removing temporary repository $TESTTMP/test-repo/.hg/upgrade.* (glob)
               $ ls -1 .hg/store/ | egrep '00(changelog|manifest)(\.n|-.*\.nd)'
 changelog-*.nd (glob)
 changelog.n
 manifest-*.nd (glob)
 manifest.n
               undo.backup.00changelog.n
               undo.backup.00manifest.n
               $ hg debugnodemap --metadata
               uid: * (glob)
               tip-rev: 5005
               tip-node: 90d5d3ba2fc47db50f712570487cb261a68c8ffe
               data-length: 121088
               data-unused: 0
               data-unused: 0.000%
             Running unrelated upgrade
               $ hg debugupgraderepo --run --no-backup --quiet --optimize re-delta-all
               upgrade will perform the following actions:
               requirements
                  preserved: dotencode, fncache, generaldelta, persistent-nodemap, revlogv1, sparserevlog, store (no-zstd !)
                  preserved: dotencode, fncache, generaldelta, persistent-nodemap, revlog-compression-zstd, revlogv1, sparserevlog, store (zstd !)
               optimisations: re-delta-all
               processed revlogs:
                 - all-filelogs
                 - changelog
                 - manifest
               $ ls -1 .hg/store/ | egrep '00(changelog|manifest)(\.n|-.*\.nd)'
 changelog-*.nd (glob)
 changelog.n
 manifest-*.nd (glob)
 manifest.n
               $ hg debugnodemap --metadata
               uid: * (glob)
               tip-rev: 5005
               tip-node: 90d5d3ba2fc47db50f712570487cb261a68c8ffe
               data-length: 121088
               data-unused: 0
               data-unused: 0.000%
             Persistent nodemap and local/streaming clone
             ============================================
               $ cd ..
             standard clone
             --------------
             The persistent nodemap should exist after a streaming clone
               $ hg clone --pull --quiet -U test-repo standard-clone
               $ ls -1 standard-clone/.hg/store/ | egrep '00(changelog|manifest)(\.n|-.*\.nd)'
 changelog-*.nd (glob)
 changelog.n
 manifest-*.nd (glob)
 manifest.n
               $ hg -R standard-clone debugnodemap --metadata
               uid: * (glob)
               tip-rev: 5005
               tip-node: 90d5d3ba2fc47db50f712570487cb261a68c8ffe
               data-length: 121088
               data-unused: 0
               data-unused: 0.000%
             local clone
             ------------
             The persistent nodemap should exist after a streaming clone
               $ hg clone -U test-repo local-clone
               $ ls -1 local-clone/.hg/store/ | egrep '00(changelog|manifest)(\.n|-.*\.nd)'
 changelog-*.nd (glob)
 changelog.n
 manifest-*.nd (glob)
 manifest.n
               $ hg -R local-clone debugnodemap --metadata
               uid: * (glob)
               tip-rev: 5005
               tip-node: 90d5d3ba2fc47db50f712570487cb261a68c8ffe
               data-length: 121088
               data-unused: 0
               data-unused: 0.000%
             stream clone
             ------------
             The persistent nodemap should exist after a streaming clone
               $ hg clone -U --stream --config ui.ssh="\"$PYTHON\" \"$TESTDIR/dummyssh\"" ssh://user@dummy/test-repo stream-clone --debug | egrep '00(changelog|manifest)'
               adding [s] 00manifest.n (70 bytes)
-              adding [s] 00manifest.i (313 KB)
               adding [s] 00manifest.d (452 KB) (no-zstd !)
               adding [s] 00manifest.d (491 KB) (zstd !)
               adding [s] 00manifest-*.nd (118 KB) (glob)
               adding [s] 00changelog.n (70 bytes)
-              adding [s] 00changelog.i (313 KB)
               adding [s] 00changelog.d (360 KB) (no-zstd !)
               adding [s] 00changelog.d (368 KB) (zstd !)
               adding [s] 00changelog-*.nd (118 KB) (glob)
+              adding [s] 00manifest.i (313 KB)
+              adding [s] 00changelog.i (313 KB)
               $ ls -1 stream-clone/.hg/store/ | egrep '00(changelog|manifest)(\.n|-.*\.nd)'
 changelog-*.nd (glob)
 changelog.n
 manifest-*.nd (glob)
 manifest.n
               $ hg -R stream-clone debugnodemap --metadata
               uid: * (glob)
               tip-rev: 5005
               tip-node: 90d5d3ba2fc47db50f712570487cb261a68c8ffe
               data-length: 121088
               data-unused: 0
               data-unused: 0.000%

General Comments 0

Write
Preview

You need to be logged in to leave comments. Login now

No TODOs yet

	Site-wide shortcuts
/	Use quick search box
g h	Goto home page
g g	Goto my private gists page
g G	Goto my public gists page
g 0-9	Goto bookmarked items from 0-9
n r	New repository page
n g	New gist page

	Repositories
g s	Goto summary page
g c	Goto changelog page
g f	Goto files page
g F	Goto files page with file search activated
g p	Goto pull requests page
g o	Goto repository settings
g O	Goto repository access permissions settings
t s	Toggle sidebar on some pages