upstream/mercurial-mirror Commit - r51365:9fdc28e2

store: introduce a EntryFile object to actually access file info...

marmoute -

r51365:9fdc28e2 default

parent child

hgext/narrow/narrowcommands.py

0 +4 -2

              # narrowcommands.py - command modifications for narrowhg extension
              #
              # Copyright 2017 Google, Inc.
              #
              # This software may be used and distributed according to the terms of the
              # GNU General Public License version 2 or any later version.
              import itertools
              import os
              from mercurial.i18n import _
              from mercurial.node import (
                  hex,
                  short,
              )
              from mercurial import (
                  bundle2,
                  cmdutil,
                  commands,
                  discovery,
                  encoding,
                  error,
                  exchange,
                  extensions,
                  hg,
                  narrowspec,
                  pathutil,
                  pycompat,
                  registrar,
                  repair,
                  repoview,
                  requirements,
                  sparse,
                  util,
                  wireprototypes,
              )
              from mercurial.utils import (
                  urlutil,
              )
              table = {}
              command = registrar.command(table)
              def setup():
                  """Wraps user-facing mercurial commands with narrow-aware versions."""
                  entry = extensions.wrapcommand(commands.table, b'clone', clonenarrowcmd)
                  entry[1].append(
                      (b'', b'narrow', None, _(b"create a narrow clone of select files"))
                  )
                  entry[1].append(
                      (
                          b'',
                          b'depth',
                          b'',
                          _(b"limit the history fetched by distance from heads"),
                      )
                  )
                  entry[1].append((b'', b'narrowspec', b'', _(b"read narrowspecs from file")))
                  # TODO(durin42): unify sparse/narrow --include/--exclude logic a bit
                  if b'sparse' not in extensions.enabled():
                      entry[1].append(
                          (b'', b'include', [], _(b"specifically fetch this file/directory"))
                      )
                      entry[1].append(
                          (
                              b'',
                              b'exclude',
                              [],
                              _(b"do not fetch this file/directory, even if included"),
                          )
                      )
                  entry = extensions.wrapcommand(commands.table, b'pull', pullnarrowcmd)
                  entry[1].append(
                      (
                          b'',
                          b'depth',
                          b'',
                          _(b"limit the history fetched by distance from heads"),
                      )
                  )
                  extensions.wrapcommand(commands.table, b'archive', archivenarrowcmd)
              def clonenarrowcmd(orig, ui, repo, *args, **opts):
                  """Wraps clone command, so 'hg clone' first wraps localrepo.clone()."""
                  opts = pycompat.byteskwargs(opts)
                  wrappedextraprepare = util.nullcontextmanager()
                  narrowspecfile = opts[b'narrowspec']
                  if narrowspecfile:
                      filepath = os.path.join(encoding.getcwd(), narrowspecfile)
                      ui.status(_(b"reading narrowspec from '%s'\n") % filepath)
                      try:
                          fdata = util.readfile(filepath)
                      except IOError as inst:
                          raise error.Abort(
                              _(b"cannot read narrowspecs from '%s': %s")
                              % (filepath, encoding.strtolocal(inst.strerror))
                          )
                      includes, excludes, profiles = sparse.parseconfig(ui, fdata, b'narrow')
                      if profiles:
                          raise error.ConfigError(
                              _(
                                  b"cannot specify other files using '%include' in"
                                  b" narrowspec"
                              )
                          )
                      narrowspec.validatepatterns(includes)
                      narrowspec.validatepatterns(excludes)
                      # narrowspec is passed so we should assume that user wants narrow clone
                      opts[b'narrow'] = True
                      opts[b'include'].extend(includes)
                      opts[b'exclude'].extend(excludes)
                  if opts[b'narrow']:
                      def pullbundle2extraprepare_widen(orig, pullop, kwargs):
                          orig(pullop, kwargs)
                          if opts.get(b'depth'):
                              kwargs[b'depth'] = opts[b'depth']
                      wrappedextraprepare = extensions.wrappedfunction(
                          exchange, b'_pullbundle2extraprepare', pullbundle2extraprepare_widen
                      )
                  with wrappedextraprepare:
                      return orig(ui, repo, *args, **pycompat.strkwargs(opts))
              def pullnarrowcmd(orig, ui, repo, *args, **opts):
                  """Wraps pull command to allow modifying narrow spec."""
                  wrappedextraprepare = util.nullcontextmanager()
                  if requirements.NARROW_REQUIREMENT in repo.requirements:
                      def pullbundle2extraprepare_widen(orig, pullop, kwargs):
                          orig(pullop, kwargs)
                          if opts.get('depth'):
                              kwargs[b'depth'] = opts['depth']
                      wrappedextraprepare = extensions.wrappedfunction(
                          exchange, b'_pullbundle2extraprepare', pullbundle2extraprepare_widen
                      )
                  with wrappedextraprepare:
                      return orig(ui, repo, *args, **opts)
              def archivenarrowcmd(orig, ui, repo, *args, **opts):
                  """Wraps archive command to narrow the default includes."""
                  if requirements.NARROW_REQUIREMENT in repo.requirements:
                      repo_includes, repo_excludes = repo.narrowpats
                      includes = set(opts.get('include', []))
                      excludes = set(opts.get('exclude', []))
                      includes, excludes, unused_invalid = narrowspec.restrictpatterns(
                          includes, excludes, repo_includes, repo_excludes
                      )
                      if includes:
                          opts['include'] = includes
                      if excludes:
                          opts['exclude'] = excludes
                  return orig(ui, repo, *args, **opts)
              def pullbundle2extraprepare(orig, pullop, kwargs):
                  repo = pullop.repo
                  if requirements.NARROW_REQUIREMENT not in repo.requirements:
                      return orig(pullop, kwargs)
                  if wireprototypes.NARROWCAP not in pullop.remote.capabilities():
                      raise error.Abort(_(b"server does not support narrow clones"))
                  orig(pullop, kwargs)
                  kwargs[b'narrow'] = True
                  include, exclude = repo.narrowpats
                  kwargs[b'oldincludepats'] = include
                  kwargs[b'oldexcludepats'] = exclude
                  if include:
                      kwargs[b'includepats'] = include
                  if exclude:
                      kwargs[b'excludepats'] = exclude
                  # calculate known nodes only in ellipses cases because in non-ellipses cases
                  # we have all the nodes
                  if wireprototypes.ELLIPSESCAP1 in pullop.remote.capabilities():
                      kwargs[b'known'] = [
                          hex(ctx.node())
                          for ctx in repo.set(b'::%ln', pullop.common)
                          if ctx.node() != repo.nullid
                      ]
                      if not kwargs[b'known']:
                          # Mercurial serializes an empty list as '' and deserializes it as
                          # [''], so delete it instead to avoid handling the empty string on
                          # the server.
                          del kwargs[b'known']
              extensions.wrapfunction(
                  exchange, b'_pullbundle2extraprepare', pullbundle2extraprepare
              )
              def _narrow(
                  ui,
                  repo,
                  remote,
                  commoninc,
                  oldincludes,
                  oldexcludes,
                  newincludes,
                  newexcludes,
                  force,
                  backup,
              ):
                  oldmatch = narrowspec.match(repo.root, oldincludes, oldexcludes)
                  newmatch = narrowspec.match(repo.root, newincludes, newexcludes)
                  # This is essentially doing "hg outgoing" to find all local-only
                  # commits. We will then check that the local-only commits don't
                  # have any changes to files that will be untracked.
                  unfi = repo.unfiltered()
                  outgoing = discovery.findcommonoutgoing(unfi, remote, commoninc=commoninc)
                  ui.status(_(b'looking for local changes to affected paths\n'))
                  progress = ui.makeprogress(
                      topic=_(b'changesets'),
                      unit=_(b'changesets'),
                      total=len(outgoing.missing) + len(outgoing.excluded),
                  )
                  localnodes = []
                  with progress:
                      for n in itertools.chain(outgoing.missing, outgoing.excluded):
                          progress.increment()
                          if any(oldmatch(f) and not newmatch(f) for f in unfi[n].files()):
                              localnodes.append(n)
                  revstostrip = unfi.revs(b'descendants(%ln)', localnodes)
                  hiddenrevs = repoview.filterrevs(repo, b'visible')
                  visibletostrip = list(
                      repo.changelog.node(r) for r in (revstostrip - hiddenrevs)
                  )
                  if visibletostrip:
                      ui.status(
                          _(
                              b'The following changeset(s) or their ancestors have '
                              b'local changes not on the remote:\n'
                          )
                      )
                      maxnodes = 10
                      if ui.verbose or len(visibletostrip) <= maxnodes:
                          for n in visibletostrip:
                              ui.status(b'%s\n' % short(n))
                      else:
                          for n in visibletostrip[:maxnodes]:
                              ui.status(b'%s\n' % short(n))
                          ui.status(
                              _(b'...and %d more, use --verbose to list all\n')
                              % (len(visibletostrip) - maxnodes)
                          )
                      if not force:
                          raise error.StateError(
                              _(b'local changes found'),
                              hint=_(b'use --force-delete-local-changes to ignore'),
                          )
                  with ui.uninterruptible():
                      if revstostrip:
                          tostrip = [unfi.changelog.node(r) for r in revstostrip]
                          if repo[b'.'].node() in tostrip:
                              # stripping working copy, so move to a different commit first
                              urev = max(
                                  repo.revs(
                                      b'(::%n) - %ln + null',
                                      repo[b'.'].node(),
                                      visibletostrip,
                                  )
                              )
                              hg.clean(repo, urev)
                          overrides = {(b'devel', b'strip-obsmarkers'): False}
                          if backup:
                              ui.status(_(b'moving unwanted changesets to backup\n'))
                          else:
                              ui.status(_(b'deleting unwanted changesets\n'))
                          with ui.configoverride(overrides, b'narrow'):
                              repair.strip(ui, unfi, tostrip, topic=b'narrow', backup=backup)
                      todelete = []
                      for entry in repo.store.datafiles():
                          f = entry.unencoded_path
                          if f.startswith(b'data/'):
                              file = f[5:-2]
                              if not newmatch(file):
-                                 todelete.append(f)
+                                 for file_ in entry.files():
+                                     todelete.append(file_.unencoded_path)
                          elif f.startswith(b'meta/'):
                              dir = f[5:-13]
                              dirs = sorted(pathutil.dirs({dir})) + [dir]
                              include = True
                              for d in dirs:
                                  visit = newmatch.visitdir(d)
                                  if not visit:
                                      include = False
                                      break
                                  if visit == b'all':
                                      break
                              if not include:
-                                 todelete.append(f)
+                                 for file_ in entry.files():
+                                     todelete.append(file_.unencoded_path)
                      repo.destroying()
                      with repo.transaction(b'narrowing'):
                          # Update narrowspec before removing revlogs, so repo won't be
                          # corrupt in case of crash
                          repo.setnarrowpats(newincludes, newexcludes)
                          for f in todelete:
                              ui.status(_(b'deleting %s\n') % f)
                              util.unlinkpath(repo.svfs.join(f))
                              repo.store.markremoved(f)
                          ui.status(_(b'deleting unwanted files from working copy\n'))
                          with repo.dirstate.changing_parents(repo):
                              narrowspec.updateworkingcopy(repo, assumeclean=True)
                              narrowspec.copytoworkingcopy(repo)
                      repo.destroyed()
              def _widen(
                  ui,
                  repo,
                  remote,
                  commoninc,
                  oldincludes,
                  oldexcludes,
                  newincludes,
                  newexcludes,
              ):
                  # for now we assume that if a server has ellipses enabled, we will be
                  # exchanging ellipses nodes. In future we should add ellipses as a client
                  # side requirement (maybe) to distinguish a client is shallow or not and
                  # then send that information to server whether we want ellipses or not.
                  # Theoretically a non-ellipses repo should be able to use narrow
                  # functionality from an ellipses enabled server
                  remotecap = remote.capabilities()
                  ellipsesremote = any(
                      cap in remotecap for cap in wireprototypes.SUPPORTED_ELLIPSESCAP
                  )
                  # check whether we are talking to a server which supports old version of
                  # ellipses capabilities
                  isoldellipses = (
                      ellipsesremote
                      and wireprototypes.ELLIPSESCAP1 in remotecap
                      and wireprototypes.ELLIPSESCAP not in remotecap
                  )
                  def pullbundle2extraprepare_widen(orig, pullop, kwargs):
                      orig(pullop, kwargs)
                      # The old{in,ex}cludepats have already been set by orig()
                      kwargs[b'includepats'] = newincludes
                      kwargs[b'excludepats'] = newexcludes
                  wrappedextraprepare = extensions.wrappedfunction(
                      exchange, b'_pullbundle2extraprepare', pullbundle2extraprepare_widen
                  )
                  # define a function that narrowbundle2 can call after creating the
                  # backup bundle, but before applying the bundle from the server
                  def setnewnarrowpats():
                      repo.setnarrowpats(newincludes, newexcludes)
                  repo.setnewnarrowpats = setnewnarrowpats
                  # silence the devel-warning of applying an empty changegroup
                  overrides = {(b'devel', b'all-warnings'): False}
                  common = commoninc[0]
                  with ui.uninterruptible():
                      if ellipsesremote:
                          ds = repo.dirstate
                          p1, p2 = ds.p1(), ds.p2()
                          with ds.changing_parents(repo):
                              ds.setparents(repo.nullid, repo.nullid)
                      if isoldellipses:
                          with wrappedextraprepare:
                              exchange.pull(repo, remote, heads=common)
                      else:
                          known = []
                          if ellipsesremote:
                              known = [
                                  ctx.node()
                                  for ctx in repo.set(b'::%ln', common)
                                  if ctx.node() != repo.nullid
                              ]
                          with remote.commandexecutor() as e:
                              bundle = e.callcommand(
                                  b'narrow_widen',
                                  {
                                      b'oldincludes': oldincludes,
                                      b'oldexcludes': oldexcludes,
                                      b'newincludes': newincludes,
                                      b'newexcludes': newexcludes,
                                      b'cgversion': b'03',
                                      b'commonheads': common,
                                      b'known': known,
                                      b'ellipses': ellipsesremote,
                                  },
                              ).result()
                          trmanager = exchange.transactionmanager(
                              repo, b'widen', remote.url()
                          )
                          with trmanager, repo.ui.configoverride(overrides, b'widen'):
                              op = bundle2.bundleoperation(
                                  repo, trmanager.transaction, source=b'widen'
                              )
                              # TODO: we should catch error.Abort here
                              bundle2.processbundle(repo, bundle, op=op, remote=remote)
                      if ellipsesremote:
                          with ds.changing_parents(repo):
                              ds.setparents(p1, p2)
                      with repo.transaction(b'widening'), repo.dirstate.changing_parents(
                          repo
                      ):
                          repo.setnewnarrowpats()
                          narrowspec.updateworkingcopy(repo)
                          narrowspec.copytoworkingcopy(repo)
              # TODO(rdamazio): Make new matcher format and update description
              @command(
                  b'tracked',
                  [
                      (b'', b'addinclude', [], _(b'new paths to include')),
                      (b'', b'removeinclude', [], _(b'old paths to no longer include')),
                      (
                          b'',
                          b'auto-remove-includes',
                          False,
                          _(b'automatically choose unused includes to remove'),
                      ),
                      (b'', b'addexclude', [], _(b'new paths to exclude')),
                      (b'', b'import-rules', b'', _(b'import narrowspecs from a file')),
                      (b'', b'removeexclude', [], _(b'old paths to no longer exclude')),
                      (
                          b'',
                          b'clear',
                          False,
                          _(b'whether to replace the existing narrowspec'),
                      ),
                      (
                          b'',
                          b'force-delete-local-changes',
                          False,
                          _(b'forces deletion of local changes when narrowing'),
                      ),
                      (
                          b'',
                          b'backup',
                          True,
                          _(b'back up local changes when narrowing'),
                      ),
                      (
                          b'',
                          b'update-working-copy',
                          False,
                          _(b'update working copy when the store has changed'),
                      ),
                  ]
                  + commands.remoteopts,
                  _(b'[OPTIONS]... [REMOTE]'),
                  inferrepo=True,
                  helpcategory=command.CATEGORY_MAINTENANCE,
              )
              def trackedcmd(ui, repo, remotepath=None, *pats, **opts):
                  """show or change the current narrowspec
                  With no argument, shows the current narrowspec entries, one per line. Each
                  line will be prefixed with 'I' or 'X' for included or excluded patterns,
                  respectively.
                  The narrowspec is comprised of expressions to match remote files and/or
                  directories that should be pulled into your client.
                  The narrowspec has *include* and *exclude* expressions, with excludes always
                  trumping includes: that is, if a file matches an exclude expression, it will
                  be excluded even if it also matches an include expression.
                  Excluding files that were never included has no effect.
                  Each included or excluded entry is in the format described by
                  'hg help patterns'.
                  The options allow you to add or remove included and excluded expressions.
                  If --clear is specified, then all previous includes and excludes are DROPPED
                  and replaced by the new ones specified to --addinclude and --addexclude.
                  If --clear is specified without any further options, the narrowspec will be
                  empty and will not match any files.
                  If --auto-remove-includes is specified, then those includes that don't match
                  any files modified by currently visible local commits (those not shared by
                  the remote) will be added to the set of explicitly specified includes to
                  remove.
                  --import-rules accepts a path to a file containing rules, allowing you to
                  add --addinclude, --addexclude rules in bulk. Like the other include and
                  exclude switches, the changes are applied immediately.
                  """
                  opts = pycompat.byteskwargs(opts)
                  if requirements.NARROW_REQUIREMENT not in repo.requirements:
                      raise error.InputError(
                          _(
                              b'the tracked command is only supported on '
                              b'repositories cloned with --narrow'
                          )
                      )
                  # Before supporting, decide whether it "hg tracked --clear" should mean
                  # tracking no paths or all paths.
                  if opts[b'clear']:
                      raise error.InputError(_(b'the --clear option is not yet supported'))
                  # import rules from a file
                  newrules = opts.get(b'import_rules')
                  if newrules:
                      try:
                          filepath = os.path.join(encoding.getcwd(), newrules)
                          fdata = util.readfile(filepath)
                      except IOError as inst:
                          raise error.StorageError(
                              _(b"cannot read narrowspecs from '%s': %s")
                              % (filepath, encoding.strtolocal(inst.strerror))
                          )
                      includepats, excludepats, profiles = sparse.parseconfig(
                          ui, fdata, b'narrow'
                      )
                      if profiles:
                          raise error.InputError(
                              _(
                                  b"including other spec files using '%include' "
                                  b"is not supported in narrowspec"
                              )
                          )
                      opts[b'addinclude'].extend(includepats)
                      opts[b'addexclude'].extend(excludepats)
                  addedincludes = narrowspec.parsepatterns(opts[b'addinclude'])
                  removedincludes = narrowspec.parsepatterns(opts[b'removeinclude'])
                  addedexcludes = narrowspec.parsepatterns(opts[b'addexclude'])
                  removedexcludes = narrowspec.parsepatterns(opts[b'removeexclude'])
                  autoremoveincludes = opts[b'auto_remove_includes']
                  update_working_copy = opts[b'update_working_copy']
                  only_show = not (
                      addedincludes
                      or removedincludes
                      or addedexcludes
                      or removedexcludes
                      or newrules
                      or autoremoveincludes
                      or update_working_copy
                  )
                  # Only print the current narrowspec.
                  if only_show:
                      oldincludes, oldexcludes = repo.narrowpats
                      ui.pager(b'tracked')
                      fm = ui.formatter(b'narrow', opts)
                      for i in sorted(oldincludes):
                          fm.startitem()
                          fm.write(b'status', b'%s ', b'I', label=b'narrow.included')
                          fm.write(b'pat', b'%s\n', i, label=b'narrow.included')
                      for i in sorted(oldexcludes):
                          fm.startitem()
                          fm.write(b'status', b'%s ', b'X', label=b'narrow.excluded')
                          fm.write(b'pat', b'%s\n', i, label=b'narrow.excluded')
                      fm.end()
                      return 0
                  with repo.wlock(), repo.lock():
                      oldincludes, oldexcludes = repo.narrowpats
                      # filter the user passed additions and deletions into actual additions and
                      # deletions of excludes and includes
                      addedincludes -= oldincludes
                      removedincludes &= oldincludes
                      addedexcludes -= oldexcludes
                      removedexcludes &= oldexcludes
                      widening = addedincludes or removedexcludes
                      narrowing = removedincludes or addedexcludes
                      if update_working_copy:
                          with repo.transaction(b'narrow-wc'), repo.dirstate.changing_parents(
                              repo
                          ):
                              narrowspec.updateworkingcopy(repo)
                              narrowspec.copytoworkingcopy(repo)
                          return 0
                      if not (widening or narrowing or autoremoveincludes):
                          ui.status(_(b"nothing to widen or narrow\n"))
                          return 0
                      cmdutil.bailifchanged(repo)
                      # Find the revisions we have in common with the remote. These will
                      # be used for finding local-only changes for narrowing. They will
                      # also define the set of revisions to update for widening.
                      path = urlutil.get_unique_pull_path_obj(b'tracked', ui, remotepath)
                      ui.status(_(b'comparing with %s\n') % urlutil.hidepassword(path.loc))
                      remote = hg.peer(repo, opts, path)
                      try:
                          # check narrow support before doing anything if widening needs to be
                          # performed. In future we should also abort if client is ellipses and
                          # server does not support ellipses
                          if (
                              widening
                              and wireprototypes.NARROWCAP not in remote.capabilities()
                          ):
                              raise error.Abort(_(b"server does not support narrow clones"))
                          commoninc = discovery.findcommonincoming(repo, remote)
                          if autoremoveincludes:
                              outgoing = discovery.findcommonoutgoing(
                                  repo, remote, commoninc=commoninc
                              )
                              ui.status(_(b'looking for unused includes to remove\n'))
                              localfiles = set()
                              for n in itertools.chain(outgoing.missing, outgoing.excluded):
                                  localfiles.update(repo[n].files())
                              suggestedremovals = []
                              for include in sorted(oldincludes):
                                  match = narrowspec.match(repo.root, [include], oldexcludes)
                                  if not any(match(f) for f in localfiles):
                                      suggestedremovals.append(include)
                              if suggestedremovals:
                                  for s in suggestedremovals:
                                      ui.status(b'%s\n' % s)
                                  if (
                                      ui.promptchoice(
                                          _(
                                              b'remove these unused includes (yn)?'
                                              b'$$ &Yes $$ &No'
                                          )
                                      )
                                      == 0
                                  ):
                                      removedincludes.update(suggestedremovals)
                                      narrowing = True
                              else:
                                  ui.status(_(b'found no unused includes\n'))
                          if narrowing:
                              newincludes = oldincludes - removedincludes
                              newexcludes = oldexcludes | addedexcludes
                              _narrow(
                                  ui,
                                  repo,
                                  remote,
                                  commoninc,
                                  oldincludes,
                                  oldexcludes,
                                  newincludes,
                                  newexcludes,
                                  opts[b'force_delete_local_changes'],
                                  opts[b'backup'],
                              )
                              # _narrow() updated the narrowspec and _widen() below needs to
                              # use the updated values as its base (otherwise removed includes
                              # and addedexcludes will be lost in the resulting narrowspec)
                              oldincludes = newincludes
                              oldexcludes = newexcludes
                          if widening:
                              newincludes = oldincludes | addedincludes
                              newexcludes = oldexcludes - removedexcludes
                              _widen(
                                  ui,
                                  repo,
                                  remote,
                                  commoninc,
                                  oldincludes,
                                  oldexcludes,
                                  newincludes,
                                  newexcludes,
                              )
                      finally:
                          remote.close()
                  return 0

mercurial/store.py

0 +18 0

              # store.py - repository store handling for Mercurial
              #
              # Copyright 2008 Olivia Mackall <olivia@selenic.com>
              #
              # This software may be used and distributed according to the terms of the
              # GNU General Public License version 2 or any later version.
              import functools
              import os
              import re
              import stat
              from typing import Generator
              from .i18n import _
              from .pycompat import getattr
              from .thirdparty import attr
              from .node import hex
              from . import (
                  changelog,
                  error,
                  manifest,
                  policy,
                  pycompat,
                  util,
                  vfs as vfsmod,
              )
              from .utils import hashutil
              parsers = policy.importmod('parsers')
              # how much bytes should be read from fncache in one read
              # It is done to prevent loading large fncache files into memory
              fncache_chunksize = 10 ** 6
              def _matchtrackedpath(path, matcher):
                  """parses a fncache entry and returns whether the entry is tracking a path
                  matched by matcher or not.
                  If matcher is None, returns True"""
                  if matcher is None:
                      return True
                  path = decodedir(path)
                  if path.startswith(b'data/'):
                      return matcher(path[len(b'data/') : -len(b'.i')])
                  elif path.startswith(b'meta/'):
                      return matcher.visitdir(path[len(b'meta/') : -len(b'/00manifest.i')])
                  raise error.ProgrammingError(b"cannot decode path %s" % path)
              # This avoids a collision between a file named foo and a dir named
              # foo.i or foo.d
              def _encodedir(path):
                  """
                  >>> _encodedir(b'data/foo.i')
                  'data/foo.i'
                  >>> _encodedir(b'data/foo.i/bla.i')
                  'data/foo.i.hg/bla.i'
                  >>> _encodedir(b'data/foo.i.hg/bla.i')
                  'data/foo.i.hg.hg/bla.i'
                  >>> _encodedir(b'data/foo.i\\ndata/foo.i/bla.i\\ndata/foo.i.hg/bla.i\\n')
                  'data/foo.i\\ndata/foo.i.hg/bla.i\\ndata/foo.i.hg.hg/bla.i\\n'
                  """
                  return (
                      path.replace(b".hg/", b".hg.hg/")
                      .replace(b".i/", b".i.hg/")
                      .replace(b".d/", b".d.hg/")
                  )
              encodedir = getattr(parsers, 'encodedir', _encodedir)
              def decodedir(path):
                  """
                  >>> decodedir(b'data/foo.i')
                  'data/foo.i'
                  >>> decodedir(b'data/foo.i.hg/bla.i')
                  'data/foo.i/bla.i'
                  >>> decodedir(b'data/foo.i.hg.hg/bla.i')
                  'data/foo.i.hg/bla.i'
                  """
                  if b".hg/" not in path:
                      return path
                  return (
                      path.replace(b".d.hg/", b".d/")
                      .replace(b".i.hg/", b".i/")
                      .replace(b".hg.hg/", b".hg/")
                  )
              def _reserved():
                  """characters that are problematic for filesystems
                  * ascii escapes (0..31)
                  * ascii hi (126..255)
                  * windows specials
                  these characters will be escaped by encodefunctions
                  """
                  winreserved = [ord(x) for x in u'\\:*?"<>|']
                  for x in range(32):
                      yield x
                  for x in range(126, 256):
                      yield x
                  for x in winreserved:
                      yield x
              def _buildencodefun():
                  """
                  >>> enc, dec = _buildencodefun()
                  >>> enc(b'nothing/special.txt')
                  'nothing/special.txt'
                  >>> dec(b'nothing/special.txt')
                  'nothing/special.txt'
                  >>> enc(b'HELLO')
                  '_h_e_l_l_o'
                  >>> dec(b'_h_e_l_l_o')
                  'HELLO'
                  >>> enc(b'hello:world?')
                  'hello~3aworld~3f'
                  >>> dec(b'hello~3aworld~3f')
                  'hello:world?'
                  >>> enc(b'the\\x07quick\\xADshot')
                  'the~07quick~adshot'
                  >>> dec(b'the~07quick~adshot')
                  'the\\x07quick\\xadshot'
                  """
                  e = b'_'
                  xchr = pycompat.bytechr
                  asciistr = list(map(xchr, range(127)))
                  capitals = list(range(ord(b"A"), ord(b"Z") + 1))
                  cmap = {x: x for x in asciistr}
                  for x in _reserved():
                      cmap[xchr(x)] = b"~%02x" % x
                  for x in capitals + [ord(e)]:
                      cmap[xchr(x)] = e + xchr(x).lower()
                  dmap = {}
                  for k, v in cmap.items():
                      dmap[v] = k
                  def decode(s):
                      i = 0
                      while i < len(s):
                          for l in range(1, 4):
                              try:
                                  yield dmap[s[i : i + l]]
                                  i += l
                                  break
                              except KeyError:
                                  pass
                          else:
                              raise KeyError
                  return (
                      lambda s: b''.join([cmap[s[c : c + 1]] for c in range(len(s))]),
                      lambda s: b''.join(list(decode(s))),
                  )
              _encodefname, _decodefname = _buildencodefun()
              def encodefilename(s):
                  """
                  >>> encodefilename(b'foo.i/bar.d/bla.hg/hi:world?/HELLO')
                  'foo.i.hg/bar.d.hg/bla.hg.hg/hi~3aworld~3f/_h_e_l_l_o'
                  """
                  return _encodefname(encodedir(s))
              def decodefilename(s):
                  """
                  >>> decodefilename(b'foo.i.hg/bar.d.hg/bla.hg.hg/hi~3aworld~3f/_h_e_l_l_o')
                  'foo.i/bar.d/bla.hg/hi:world?/HELLO'
                  """
                  return decodedir(_decodefname(s))
              def _buildlowerencodefun():
                  """
                  >>> f = _buildlowerencodefun()
                  >>> f(b'nothing/special.txt')
                  'nothing/special.txt'
                  >>> f(b'HELLO')
                  'hello'
                  >>> f(b'hello:world?')
                  'hello~3aworld~3f'
                  >>> f(b'the\\x07quick\\xADshot')
                  'the~07quick~adshot'
                  """
                  xchr = pycompat.bytechr
                  cmap = {xchr(x): xchr(x) for x in range(127)}
                  for x in _reserved():
                      cmap[xchr(x)] = b"~%02x" % x
                  for x in range(ord(b"A"), ord(b"Z") + 1):
                      cmap[xchr(x)] = xchr(x).lower()
                  def lowerencode(s):
                      return b"".join([cmap[c] for c in pycompat.iterbytestr(s)])
                  return lowerencode
              lowerencode = getattr(parsers, 'lowerencode', None) or _buildlowerencodefun()
              # Windows reserved names: con, prn, aux, nul, com1..com9, lpt1..lpt9
              _winres3 = (b'aux', b'con', b'prn', b'nul')  # length 3
              _winres4 = (b'com', b'lpt')  # length 4 (with trailing 1..9)
              def _auxencode(path, dotencode):
                  """
                  Encodes filenames containing names reserved by Windows or which end in
                  period or space. Does not touch other single reserved characters c.
                  Specifically, c in '\\:*?"<>|' or ord(c) <= 31 are *not* encoded here.
                  Additionally encodes space or period at the beginning, if dotencode is
                  True. Parameter path is assumed to be all lowercase.
                  A segment only needs encoding if a reserved name appears as a
                  basename (e.g. "aux", "aux.foo"). A directory or file named "foo.aux"
                  doesn't need encoding.
                  >>> s = b'.foo/aux.txt/txt.aux/con/prn/nul/foo.'
                  >>> _auxencode(s.split(b'/'), True)
                  ['~2efoo', 'au~78.txt', 'txt.aux', 'co~6e', 'pr~6e', 'nu~6c', 'foo~2e']
                  >>> s = b'.com1com2/lpt9.lpt4.lpt1/conprn/com0/lpt0/foo.'
                  >>> _auxencode(s.split(b'/'), False)
                  ['.com1com2', 'lp~749.lpt4.lpt1', 'conprn', 'com0', 'lpt0', 'foo~2e']
                  >>> _auxencode([b'foo. '], True)
                  ['foo.~20']
                  >>> _auxencode([b' .foo'], True)
                  ['~20.foo']
                  """
                  for i, n in enumerate(path):
                      if not n:
                          continue
                      if dotencode and n[0] in b'. ':
                          n = b"~%02x" % ord(n[0:1]) + n[1:]
                          path[i] = n
                      else:
                          l = n.find(b'.')
                          if l == -1:
                              l = len(n)
                          if (l == 3 and n[:3] in _winres3) or (
                              l == 4
                              and n[3:4] <= b'9'
                              and n[3:4] >= b'1'
                              and n[:3] in _winres4
                          ):
                              # encode third letter ('aux' -> 'au~78')
                              ec = b"~%02x" % ord(n[2:3])
                              n = n[0:2] + ec + n[3:]
                              path[i] = n
                      if n[-1] in b'. ':
                          # encode last period or space ('foo...' -> 'foo..~2e')
                          path[i] = n[:-1] + b"~%02x" % ord(n[-1:])
                  return path
              _maxstorepathlen = 120
              _dirprefixlen = 8
              _maxshortdirslen = 8 * (_dirprefixlen + 1) - 4
              def _hashencode(path, dotencode):
                  digest = hex(hashutil.sha1(path).digest())
                  le = lowerencode(path[5:]).split(b'/')  # skips prefix 'data/' or 'meta/'
                  parts = _auxencode(le, dotencode)
                  basename = parts[-1]
                  _root, ext = os.path.splitext(basename)
                  sdirs = []
                  sdirslen = 0
                  for p in parts[:-1]:
                      d = p[:_dirprefixlen]
                      if d[-1] in b'. ':
                          # Windows can't access dirs ending in period or space
                          d = d[:-1] + b'_'
                      if sdirslen == 0:
                          t = len(d)
                      else:
                          t = sdirslen + 1 + len(d)
                          if t > _maxshortdirslen:
                              break
                      sdirs.append(d)
                      sdirslen = t
                  dirs = b'/'.join(sdirs)
                  if len(dirs) > 0:
                      dirs += b'/'
                  res = b'dh/' + dirs + digest + ext
                  spaceleft = _maxstorepathlen - len(res)
                  if spaceleft > 0:
                      filler = basename[:spaceleft]
                      res = b'dh/' + dirs + filler + digest + ext
                  return res
              def _hybridencode(path, dotencode):
                  """encodes path with a length limit
                  Encodes all paths that begin with 'data/', according to the following.
                  Default encoding (reversible):
                  Encodes all uppercase letters 'X' as '_x'. All reserved or illegal
                  characters are encoded as '~xx', where xx is the two digit hex code
                  of the character (see encodefilename).
                  Relevant path components consisting of Windows reserved filenames are
                  masked by encoding the third character ('aux' -> 'au~78', see _auxencode).
                  Hashed encoding (not reversible):
                  If the default-encoded path is longer than _maxstorepathlen, a
                  non-reversible hybrid hashing of the path is done instead.
                  This encoding uses up to _dirprefixlen characters of all directory
                  levels of the lowerencoded path, but not more levels than can fit into
                  _maxshortdirslen.
                  Then follows the filler followed by the sha digest of the full path.
                  The filler is the beginning of the basename of the lowerencoded path
                  (the basename is everything after the last path separator). The filler
                  is as long as possible, filling in characters from the basename until
                  the encoded path has _maxstorepathlen characters (or all chars of the
                  basename have been taken).
                  The extension (e.g. '.i' or '.d') is preserved.
                  The string 'data/' at the beginning is replaced with 'dh/', if the hashed
                  encoding was used.
                  """
                  path = encodedir(path)
                  ef = _encodefname(path).split(b'/')
                  res = b'/'.join(_auxencode(ef, dotencode))
                  if len(res) > _maxstorepathlen:
                      res = _hashencode(path, dotencode)
                  return res
              def _pathencode(path):
                  de = encodedir(path)
                  if len(path) > _maxstorepathlen:
                      return _hashencode(de, True)
                  ef = _encodefname(de).split(b'/')
                  res = b'/'.join(_auxencode(ef, True))
                  if len(res) > _maxstorepathlen:
                      return _hashencode(de, True)
                  return res
              _pathencode = getattr(parsers, 'pathencode', _pathencode)
              def _plainhybridencode(f):
                  return _hybridencode(f, False)
              def _calcmode(vfs):
                  try:
                      # files in .hg/ will be created using this mode
                      mode = vfs.stat().st_mode
                      # avoid some useless chmods
                      if (0o777 & ~util.umask) == (0o777 & mode):
                          mode = None
                  except OSError:
                      mode = None
                  return mode
              _data = [
                  b'bookmarks',
                  b'narrowspec',
                  b'data',
                  b'meta',
                  b'00manifest.d',
                  b'00manifest.i',
                  b'00changelog.d',
                  b'00changelog.i',
                  b'phaseroots',
                  b'obsstore',
                  b'requires',
              ]
              REVLOG_FILES_MAIN_EXT = (b'.i',)
              REVLOG_FILES_OTHER_EXT = (
                  b'.idx',
                  b'.d',
                  b'.dat',
                  b'.n',
                  b'.nd',
                  b'.sda',
              )
              # files that are "volatile" and might change between listing and streaming
              #
              # note: the ".nd" file are nodemap data and won't "change" but they might be
              # deleted.
              REVLOG_FILES_VOLATILE_EXT = (b'.n', b'.nd')
              # some exception to the above matching
              #
              # XXX This is currently not in use because of issue6542
              EXCLUDED = re.compile(br'.*undo\.[^/]+\.(nd?|i)$')
              def is_revlog(f, kind, st):
                  if kind != stat.S_IFREG:
                      return None
                  return revlog_type(f)
              def revlog_type(f):
                  # XXX we need to filter `undo.` created by the transaction here, however
                  # being naive about it also filter revlog for `undo.*` files, leading to
                  # issue6542. So we no longer use EXCLUDED.
                  if f.endswith(REVLOG_FILES_MAIN_EXT):
                      return FILEFLAGS_REVLOG_MAIN
                  elif f.endswith(REVLOG_FILES_OTHER_EXT):
                      t = FILETYPE_FILELOG_OTHER
                      if f.endswith(REVLOG_FILES_VOLATILE_EXT):
                          t |= FILEFLAGS_VOLATILE
                      return t
                  return None
              # the file is part of changelog data
              FILEFLAGS_CHANGELOG = 1 << 13
              # the file is part of manifest data
              FILEFLAGS_MANIFESTLOG = 1 << 12
              # the file is part of filelog data
              FILEFLAGS_FILELOG = 1 << 11
              # file that are not directly part of a revlog
              FILEFLAGS_OTHER = 1 << 10
              # the main entry point for a revlog
              FILEFLAGS_REVLOG_MAIN = 1 << 1
              # a secondary file for a revlog
              FILEFLAGS_REVLOG_OTHER = 1 << 0
              # files that are "volatile" and might change between listing and streaming
              FILEFLAGS_VOLATILE = 1 << 20
              FILETYPE_CHANGELOG_MAIN = FILEFLAGS_CHANGELOG | FILEFLAGS_REVLOG_MAIN
              FILETYPE_CHANGELOG_OTHER = FILEFLAGS_CHANGELOG | FILEFLAGS_REVLOG_OTHER
              FILETYPE_MANIFESTLOG_MAIN = FILEFLAGS_MANIFESTLOG | FILEFLAGS_REVLOG_MAIN
              FILETYPE_MANIFESTLOG_OTHER = FILEFLAGS_MANIFESTLOG | FILEFLAGS_REVLOG_OTHER
              FILETYPE_FILELOG_MAIN = FILEFLAGS_FILELOG | FILEFLAGS_REVLOG_MAIN
              FILETYPE_FILELOG_OTHER = FILEFLAGS_FILELOG | FILEFLAGS_REVLOG_OTHER
              FILETYPE_OTHER = FILEFLAGS_OTHER
              @attr.s(slots=True)
              class StoreEntry:
                  """An entry in the store
                  This is returned by `store.walk` and represent some data in the store."""
                  unencoded_path = attr.ib()
                  is_revlog = attr.ib(default=False)
                  revlog_type = attr.ib(default=None)
                  is_revlog_main = attr.ib(default=None)
                  is_volatile = attr.ib(default=False)
                  file_size = attr.ib(default=None)
+                 def files(self):
+                     return [
+                         StoreFile(
+                             unencoded_path=self.unencoded_path,
+                             file_size=self.file_size,
+                             is_volatile=self.is_volatile,
+                         )
+                     ]
+             @attr.s(slots=True)
+             class StoreFile:
+                 """a file matching an entry"""
+                 unencoded_path = attr.ib()
+                 file_size = attr.ib()
+                 is_volatile = attr.ib(default=False)
              class basicstore:
                  '''base class for local repository stores'''
                  def __init__(self, path, vfstype):
                      vfs = vfstype(path)
                      self.path = vfs.base
                      self.createmode = _calcmode(vfs)
                      vfs.createmode = self.createmode
                      self.rawvfs = vfs
                      self.vfs = vfsmod.filtervfs(vfs, encodedir)
                      self.opener = self.vfs
                  def join(self, f):
                      return self.path + b'/' + encodedir(f)
                  def _walk(self, relpath, recurse):
                      '''yields (revlog_type, unencoded, size)'''
                      path = self.path
                      if relpath:
                          path += b'/' + relpath
                      striplen = len(self.path) + 1
                      l = []
                      if self.rawvfs.isdir(path):
                          visit = [path]
                          readdir = self.rawvfs.readdir
                          while visit:
                              p = visit.pop()
                              for f, kind, st in readdir(p, stat=True):
                                  fp = p + b'/' + f
                                  rl_type = is_revlog(f, kind, st)
                                  if rl_type is not None:
                                      n = util.pconvert(fp[striplen:])
                                      l.append((rl_type, decodedir(n), st.st_size))
                                  elif kind == stat.S_IFDIR and recurse:
                                      visit.append(fp)
                      l.sort()
                      return l
                  def changelog(self, trypending, concurrencychecker=None):
                      return changelog.changelog(
                          self.vfs,
                          trypending=trypending,
                          concurrencychecker=concurrencychecker,
                      )
                  def manifestlog(self, repo, storenarrowmatch):
                      rootstore = manifest.manifestrevlog(repo.nodeconstants, self.vfs)
                      return manifest.manifestlog(self.vfs, repo, rootstore, storenarrowmatch)
                  def datafiles(
                      self, matcher=None, undecodable=None
                  ) -> Generator[StoreEntry, None, None]:
                      """Like walk, but excluding the changelog and root manifest.
                      When [undecodable] is None, revlogs names that can't be
                      decoded cause an exception. When it is provided, it should
                      be a list and the filenames that can't be decoded are added
                      to it instead. This is very rarely needed."""
                      files = self._walk(b'data', True) + self._walk(b'meta', True)
                      for (t, u, s) in files:
                          if t is not None:
                              yield StoreEntry(
                                  unencoded_path=u,
                                  is_revlog=True,
                                  revlog_type=FILEFLAGS_FILELOG,
                                  is_revlog_main=bool(t & FILEFLAGS_REVLOG_MAIN),
                                  is_volatile=bool(t & FILEFLAGS_VOLATILE),
                                  file_size=s,
                              )
                  def topfiles(self) -> Generator[StoreEntry, None, None]:
                      # yield manifest before changelog
                      files = reversed(self._walk(b'', False))
                      for (t, u, s) in files:
                          if u.startswith(b'00changelog'):
                              revlog_type = FILEFLAGS_CHANGELOG
                          elif u.startswith(b'00manifest'):
                              revlog_type = FILEFLAGS_MANIFESTLOG
                          else:
                              revlog_type = None
                          yield StoreEntry(
                              unencoded_path=u,
                              is_revlog=revlog_type is not None,
                              revlog_type=revlog_type,
                              is_revlog_main=bool(t & FILEFLAGS_REVLOG_MAIN),
                              is_volatile=bool(t & FILEFLAGS_VOLATILE),
                              file_size=s,
                          )
                  def walk(self, matcher=None) -> Generator[StoreEntry, None, None]:
                      """return files related to data storage (ie: revlogs)
                      yields (file_type, unencoded, size)
                      if a matcher is passed, storage files of only those tracked paths
                      are passed with matches the matcher
                      """
                      # yield data files first
                      for x in self.datafiles(matcher):
                          yield x
                      for x in self.topfiles():
                          yield x
                  def copylist(self):
                      return _data
                  def write(self, tr):
                      pass
                  def invalidatecaches(self):
                      pass
                  def markremoved(self, fn):
                      pass
                  def __contains__(self, path):
                      '''Checks if the store contains path'''
                      path = b"/".join((b"data", path))
                      # file?
                      if self.vfs.exists(path + b".i"):
                          return True
                      # dir?
                      if not path.endswith(b"/"):
                          path = path + b"/"
                      return self.vfs.exists(path)
              class encodedstore(basicstore):
                  def __init__(self, path, vfstype):
                      vfs = vfstype(path + b'/store')
                      self.path = vfs.base
                      self.createmode = _calcmode(vfs)
                      vfs.createmode = self.createmode
                      self.rawvfs = vfs
                      self.vfs = vfsmod.filtervfs(vfs, encodefilename)
                      self.opener = self.vfs
                  # note: topfiles would also need a decode phase. It is just that in
                  # practice we do not have any file outside of `data/` that needs encoding.
                  # However that might change so we should probably add a test and encoding
                  # decoding for it too. see issue6548
                  def datafiles(
                      self, matcher=None, undecodable=None
                  ) -> Generator[StoreEntry, None, None]:
                      for entry in super(encodedstore, self).datafiles():
                          try:
                              f1 = entry.unencoded_path
                              f2 = decodefilename(f1)
                          except KeyError:
                              if undecodable is None:
                                  msg = _(b'undecodable revlog name %s') % f1
                                  raise error.StorageError(msg)
                              else:
                                  undecodable.append(f1)
                                  continue
                          if not _matchtrackedpath(f2, matcher):
                              continue
                          entry.unencoded_path = f2
                          yield entry
                  def join(self, f):
                      return self.path + b'/' + encodefilename(f)
                  def copylist(self):
                      return [b'requires', b'00changelog.i'] + [b'store/' + f for f in _data]
              class fncache:
                  # the filename used to be partially encoded
                  # hence the encodedir/decodedir dance
                  def __init__(self, vfs):
                      self.vfs = vfs
                      self._ignores = set()
                      self.entries = None
                      self._dirty = False
                      # set of new additions to fncache
                      self.addls = set()
                  def ensureloaded(self, warn=None):
                      """read the fncache file if not already read.
                      If the file on disk is corrupted, raise. If warn is provided,
                      warn and keep going instead."""
                      if self.entries is None:
                          self._load(warn)
                  def _load(self, warn=None):
                      '''fill the entries from the fncache file'''
                      self._dirty = False
                      try:
                          fp = self.vfs(b'fncache', mode=b'rb')
                      except IOError:
                          # skip nonexistent file
                          self.entries = set()
                          return
                      self.entries = set()
                      chunk = b''
                      for c in iter(functools.partial(fp.read, fncache_chunksize), b''):
                          chunk += c
                          try:
                              p = chunk.rindex(b'\n')
                              self.entries.update(decodedir(chunk[: p + 1]).splitlines())
                              chunk = chunk[p + 1 :]
                          except ValueError:
                              # substring '\n' not found, maybe the entry is bigger than the
                              # chunksize, so let's keep iterating
                              pass
                      if chunk:
                          msg = _(b"fncache does not ends with a newline")
                          if warn:
                              warn(msg + b'\n')
                          else:
                              raise error.Abort(
                                  msg,
                                  hint=_(
                                      b"use 'hg debugrebuildfncache' to "
                                      b"rebuild the fncache"
                                  ),
                              )
                      self._checkentries(fp, warn)
                      fp.close()
                  def _checkentries(self, fp, warn):
                      """make sure there is no empty string in entries"""
                      if b'' in self.entries:
                          fp.seek(0)
                          for n, line in enumerate(fp):
                              if not line.rstrip(b'\n'):
                                  t = _(b'invalid entry in fncache, line %d') % (n + 1)
                                  if warn:
                                      warn(t + b'\n')
                                  else:
                                      raise error.Abort(t)
                  def write(self, tr):
                      if self._dirty:
                          assert self.entries is not None
                          self.entries = self.entries | self.addls
                          self.addls = set()
                          tr.addbackup(b'fncache')
                          fp = self.vfs(b'fncache', mode=b'wb', atomictemp=True)
                          if self.entries:
                              fp.write(encodedir(b'\n'.join(self.entries) + b'\n'))
                          fp.close()
                          self._dirty = False
                      if self.addls:
                          # if we have just new entries, let's append them to the fncache
                          tr.addbackup(b'fncache')
                          fp = self.vfs(b'fncache', mode=b'ab', atomictemp=True)
                          if self.addls:
                              fp.write(encodedir(b'\n'.join(self.addls) + b'\n'))
                          fp.close()
                          self.entries = None
                          self.addls = set()
                  def addignore(self, fn):
                      self._ignores.add(fn)
                  def add(self, fn):
                      if fn in self._ignores:
                          return
                      if self.entries is None:
                          self._load()
                      if fn not in self.entries:
                          self.addls.add(fn)
                  def remove(self, fn):
                      if self.entries is None:
                          self._load()
                      if fn in self.addls:
                          self.addls.remove(fn)
                          return
                      try:
                          self.entries.remove(fn)
                          self._dirty = True
                      except KeyError:
                          pass
                  def __contains__(self, fn):
                      if fn in self.addls:
                          return True
                      if self.entries is None:
                          self._load()
                      return fn in self.entries
                  def __iter__(self):
                      if self.entries is None:
                          self._load()
                      return iter(self.entries | self.addls)
              class _fncachevfs(vfsmod.proxyvfs):
                  def __init__(self, vfs, fnc, encode):
                      vfsmod.proxyvfs.__init__(self, vfs)
                      self.fncache = fnc
                      self.encode = encode
                  def __call__(self, path, mode=b'r', *args, **kw):
                      encoded = self.encode(path)
                      if (
                          mode not in (b'r', b'rb')
                          and (path.startswith(b'data/') or path.startswith(b'meta/'))
                          and revlog_type(path) is not None
                      ):
                          # do not trigger a fncache load when adding a file that already is
                          # known to exist.
                          notload = self.fncache.entries is None and self.vfs.exists(encoded)
                          if notload and b'r+' in mode and not self.vfs.stat(encoded).st_size:
                              # when appending to an existing file, if the file has size zero,
                              # it should be considered as missing. Such zero-size files are
                              # the result of truncation when a transaction is aborted.
                              notload = False
                          if not notload:
                              self.fncache.add(path)
                      return self.vfs(encoded, mode, *args, **kw)
                  def join(self, path):
                      if path:
                          return self.vfs.join(self.encode(path))
                      else:
                          return self.vfs.join(path)
                  def register_file(self, path):
                      """generic hook point to lets fncache steer its stew"""
                      if path.startswith(b'data/') or path.startswith(b'meta/'):
                          self.fncache.add(path)
              class fncachestore(basicstore):
                  def __init__(self, path, vfstype, dotencode):
                      if dotencode:
                          encode = _pathencode
                      else:
                          encode = _plainhybridencode
                      self.encode = encode
                      vfs = vfstype(path + b'/store')
                      self.path = vfs.base
                      self.pathsep = self.path + b'/'
                      self.createmode = _calcmode(vfs)
                      vfs.createmode = self.createmode
                      self.rawvfs = vfs
                      fnc = fncache(vfs)
                      self.fncache = fnc
                      self.vfs = _fncachevfs(vfs, fnc, encode)
                      self.opener = self.vfs
                  def join(self, f):
                      return self.pathsep + self.encode(f)
                  def getsize(self, path):
                      return self.rawvfs.stat(path).st_size
                  def datafiles(
                      self, matcher=None, undecodable=None
                  ) -> Generator[StoreEntry, None, None]:
                      for f in sorted(self.fncache):
                          if not _matchtrackedpath(f, matcher):
                              continue
                          ef = self.encode(f)
                          t = revlog_type(f)
                          if t is None:
                              # Note: this should not be in the fncache then…
                              #
                              # However the fncache might contains such file added by
                              # previous version of Mercurial.
                              continue
                          t |= FILEFLAGS_FILELOG
                          try:
                              yield StoreEntry(
                                  unencoded_path=f,
                                  is_revlog=True,
                                  revlog_type=FILEFLAGS_FILELOG,
                                  is_revlog_main=bool(t & FILEFLAGS_REVLOG_MAIN),
                                  is_volatile=bool(t & FILEFLAGS_VOLATILE),
                                  file_size=self.getsize(ef),
                              )
                          except FileNotFoundError:
                              pass
                  def copylist(self):
                      d = (
                          b'bookmarks',
                          b'narrowspec',
                          b'data',
                          b'meta',
                          b'dh',
                          b'fncache',
                          b'phaseroots',
                          b'obsstore',
                          b'00manifest.d',
                          b'00manifest.i',
                          b'00changelog.d',
                          b'00changelog.i',
                          b'requires',
                      )
                      return [b'requires', b'00changelog.i'] + [b'store/' + f for f in d]
                  def write(self, tr):
                      self.fncache.write(tr)
                  def invalidatecaches(self):
                      self.fncache.entries = None
                      self.fncache.addls = set()
                  def markremoved(self, fn):
                      self.fncache.remove(fn)
                  def _exists(self, f):
                      ef = self.encode(f)
                      try:
                          self.getsize(ef)
                          return True
                      except FileNotFoundError:
                          return False
                  def __contains__(self, path):
                      '''Checks if the store contains path'''
                      path = b"/".join((b"data", path))
                      # check for files (exact match)
                      e = path + b'.i'
                      if e in self.fncache and self._exists(e):
                          return True
                      # now check for directories (prefix match)
                      if not path.endswith(b'/'):
                          path += b'/'
                      for e in self.fncache:
                          if e.startswith(path) and self._exists(e):
                              return True
                      return False

mercurial/streamclone.py

0 +9 -9

              # streamclone.py - producing and consuming streaming repository data
              #
              # Copyright 2015 Gregory Szorc <gregory.szorc@gmail.com>
              #
              # This software may be used and distributed according to the terms of the
              # GNU General Public License version 2 or any later version.
              import contextlib
              import os
              import struct
              from .i18n import _
              from .pycompat import open
              from .interfaces import repository
              from . import (
                  bookmarks,
                  cacheutil,
                  error,
                  narrowspec,
                  phases,
                  pycompat,
                  requirements as requirementsmod,
                  scmutil,
                  store,
                  transaction,
                  util,
              )
              from .revlogutils import (
                  nodemap,
              )
              def new_stream_clone_requirements(default_requirements, streamed_requirements):
                  """determine the final set of requirement for a new stream clone
                  this method combine the "default" requirements that a new repository would
                  use with the constaint we get from the stream clone content. We keep local
                  configuration choice when possible.
                  """
                  requirements = set(default_requirements)
                  requirements -= requirementsmod.STREAM_FIXED_REQUIREMENTS
                  requirements.update(streamed_requirements)
                  return requirements
              def streamed_requirements(repo):
                  """the set of requirement the new clone will have to support
                  This is used for advertising the stream options and to generate the actual
                  stream content."""
                  requiredformats = (
                      repo.requirements & requirementsmod.STREAM_FIXED_REQUIREMENTS
                  )
                  return requiredformats
              def canperformstreamclone(pullop, bundle2=False):
                  """Whether it is possible to perform a streaming clone as part of pull.
                  ``bundle2`` will cause the function to consider stream clone through
                  bundle2 and only through bundle2.
                  Returns a tuple of (supported, requirements). ``supported`` is True if
                  streaming clone is supported and False otherwise. ``requirements`` is
                  a set of repo requirements from the remote, or ``None`` if stream clone
                  isn't supported.
                  """
                  repo = pullop.repo
                  remote = pullop.remote
                  bundle2supported = False
                  if pullop.canusebundle2:
                      if b'v2' in pullop.remotebundle2caps.get(b'stream', []):
                          bundle2supported = True
                      # else
                      # Server doesn't support bundle2 stream clone or doesn't support
                      # the versions we support. Fall back and possibly allow legacy.
                  # Ensures legacy code path uses available bundle2.
                  if bundle2supported and not bundle2:
                      return False, None
                  # Ensures bundle2 doesn't try to do a stream clone if it isn't supported.
                  elif bundle2 and not bundle2supported:
                      return False, None
                  # Streaming clone only works on empty repositories.
                  if len(repo):
                      return False, None
                  # Streaming clone only works if all data is being requested.
                  if pullop.heads:
                      return False, None
                  streamrequested = pullop.streamclonerequested
                  # If we don't have a preference, let the server decide for us. This
                  # likely only comes into play in LANs.
                  if streamrequested is None:
                      # The server can advertise whether to prefer streaming clone.
                      streamrequested = remote.capable(b'stream-preferred')
                  if not streamrequested:
                      return False, None
                  # In order for stream clone to work, the client has to support all the
                  # requirements advertised by the server.
                  #
                  # The server advertises its requirements via the "stream" and "streamreqs"
                  # capability. "stream" (a value-less capability) is advertised if and only
                  # if the only requirement is "revlogv1." Else, the "streamreqs" capability
                  # is advertised and contains a comma-delimited list of requirements.
                  requirements = set()
                  if remote.capable(b'stream'):
                      requirements.add(requirementsmod.REVLOGV1_REQUIREMENT)
                  else:
                      streamreqs = remote.capable(b'streamreqs')
                      # This is weird and shouldn't happen with modern servers.
                      if not streamreqs:
                          pullop.repo.ui.warn(
                              _(
                                  b'warning: stream clone requested but server has them '
                                  b'disabled\n'
                              )
                          )
                          return False, None
                      streamreqs = set(streamreqs.split(b','))
                      # Server requires something we don't support. Bail.
                      missingreqs = streamreqs - repo.supported
                      if missingreqs:
                          pullop.repo.ui.warn(
                              _(
                                  b'warning: stream clone requested but client is missing '
                                  b'requirements: %s\n'
                              )
                              % b', '.join(sorted(missingreqs))
                          )
                          pullop.repo.ui.warn(
                              _(
                                  b'(see https://www.mercurial-scm.org/wiki/MissingRequirement '
                                  b'for more information)\n'
                              )
                          )
                          return False, None
                      requirements = streamreqs
                  return True, requirements
              def maybeperformlegacystreamclone(pullop):
                  """Possibly perform a legacy stream clone operation.
                  Legacy stream clones are performed as part of pull but before all other
                  operations.
                  A legacy stream clone will not be performed if a bundle2 stream clone is
                  supported.
                  """
                  from . import localrepo
                  supported, requirements = canperformstreamclone(pullop)
                  if not supported:
                      return
                  repo = pullop.repo
                  remote = pullop.remote
                  # Save remote branchmap. We will use it later to speed up branchcache
                  # creation.
                  rbranchmap = None
                  if remote.capable(b'branchmap'):
                      with remote.commandexecutor() as e:
                          rbranchmap = e.callcommand(b'branchmap', {}).result()
                  repo.ui.status(_(b'streaming all changes\n'))
                  with remote.commandexecutor() as e:
                      fp = e.callcommand(b'stream_out', {}).result()
                  # TODO strictly speaking, this code should all be inside the context
                  # manager because the context manager is supposed to ensure all wire state
                  # is flushed when exiting. But the legacy peers don't do this, so it
                  # doesn't matter.
                  l = fp.readline()
                  try:
                      resp = int(l)
                  except ValueError:
                      raise error.ResponseError(
                          _(b'unexpected response from remote server:'), l
                      )
                  if resp == 1:
                      raise error.Abort(_(b'operation forbidden by server'))
                  elif resp == 2:
                      raise error.Abort(_(b'locking the remote repository failed'))
                  elif resp != 0:
                      raise error.Abort(_(b'the server sent an unknown error code'))
                  l = fp.readline()
                  try:
                      filecount, bytecount = map(int, l.split(b' ', 1))
                  except (ValueError, TypeError):
                      raise error.ResponseError(
                          _(b'unexpected response from remote server:'), l
                      )
                  with repo.lock():
                      consumev1(repo, fp, filecount, bytecount)
                      repo.requirements = new_stream_clone_requirements(
                          repo.requirements,
                          requirements,
                      )
                      repo.svfs.options = localrepo.resolvestorevfsoptions(
                          repo.ui, repo.requirements, repo.features
                      )
                      scmutil.writereporequirements(repo)
                      nodemap.post_stream_cleanup(repo)
                      if rbranchmap:
                          repo._branchcaches.replace(repo, rbranchmap)
                      repo.invalidate()
              def allowservergeneration(repo):
                  """Whether streaming clones are allowed from the server."""
                  if repository.REPO_FEATURE_STREAM_CLONE not in repo.features:
                      return False
                  if not repo.ui.configbool(b'server', b'uncompressed', untrusted=True):
                      return False
                  # The way stream clone works makes it impossible to hide secret changesets.
                  # So don't allow this by default.
                  secret = phases.hassecret(repo)
                  if secret:
                      return repo.ui.configbool(b'server', b'uncompressedallowsecret')
                  return True
              # This is it's own function so extensions can override it.
              def _walkstreamfiles(repo, matcher=None):
                  return repo.store.walk(matcher)
              def generatev1(repo):
                  """Emit content for version 1 of a streaming clone.
                  This returns a 3-tuple of (file count, byte size, data iterator).
                  The data iterator consists of N entries for each file being transferred.
                  Each file entry starts as a line with the file name and integer size
                  delimited by a null byte.
                  The raw file data follows. Following the raw file data is the next file
                  entry, or EOF.
                  When used on the wire protocol, an additional line indicating protocol
                  success will be prepended to the stream. This function is not responsible
                  for adding it.
                  This function will obtain a repository lock to ensure a consistent view of
                  the store is captured. It therefore may raise LockError.
                  """
                  entries = []
                  total_bytes = 0
                  # Get consistent snapshot of repo, lock during scan.
                  with repo.lock():
                      repo.ui.debug(b'scanning\n')
                      for entry in _walkstreamfiles(repo):
-                         if entry.file_size:
-                             entries.append((entry.unencoded_path, entry.file_size))
-                             total_bytes += entry.file_size
+                         for f in entry.files():
+                             if f.file_size:
+                                 entries.append((f.unencoded_path, f.file_size))
+                                 total_bytes += f.file_size
                      _test_sync_point_walk_1(repo)
                  _test_sync_point_walk_2(repo)
                  repo.ui.debug(
                      b'%d files, %d bytes to transfer\n' % (len(entries), total_bytes)
                  )
                  svfs = repo.svfs
                  debugflag = repo.ui.debugflag
                  def emitrevlogdata():
                      for name, size in entries:
                          if debugflag:
                              repo.ui.debug(b'sending %s (%d bytes)\n' % (name, size))
                          # partially encode name over the wire for backwards compat
                          yield b'%s\0%d\n' % (store.encodedir(name), size)
                          # auditing at this stage is both pointless (paths are already
                          # trusted by the local repo) and expensive
                          with svfs(name, b'rb', auditpath=False) as fp:
                              if size <= 65536:
                                  yield fp.read(size)
                              else:
                                  for chunk in util.filechunkiter(fp, limit=size):
                                      yield chunk
                  return len(entries), total_bytes, emitrevlogdata()
              def generatev1wireproto(repo):
                  """Emit content for version 1 of streaming clone suitable for the wire.
                  This is the data output from ``generatev1()`` with 2 header lines. The
                  first line indicates overall success. The 2nd contains the file count and
                  byte size of payload.
                  The success line contains "0" for success, "1" for stream generation not
                  allowed, and "2" for error locking the repository (possibly indicating
                  a permissions error for the server process).
                  """
                  if not allowservergeneration(repo):
                      yield b'1\n'
                      return
                  try:
                      filecount, bytecount, it = generatev1(repo)
                  except error.LockError:
                      yield b'2\n'
                      return
                  # Indicates successful response.
                  yield b'0\n'
                  yield b'%d %d\n' % (filecount, bytecount)
                  for chunk in it:
                      yield chunk
              def generatebundlev1(repo, compression=b'UN'):
                  """Emit content for version 1 of a stream clone bundle.
                  The first 4 bytes of the output ("HGS1") denote this as stream clone
                  bundle version 1.
                  The next 2 bytes indicate the compression type. Only "UN" is currently
                  supported.
                  The next 16 bytes are two 64-bit big endian unsigned integers indicating
                  file count and byte count, respectively.
                  The next 2 bytes is a 16-bit big endian unsigned short declaring the length
                  of the requirements string, including a trailing \0. The following N bytes
                  are the requirements string, which is ASCII containing a comma-delimited
                  list of repo requirements that are needed to support the data.
                  The remaining content is the output of ``generatev1()`` (which may be
                  compressed in the future).
                  Returns a tuple of (requirements, data generator).
                  """
                  if compression != b'UN':
                      raise ValueError(b'we do not support the compression argument yet')
                  requirements = streamed_requirements(repo)
                  requires = b','.join(sorted(requirements))
                  def gen():
                      yield b'HGS1'
                      yield compression
                      filecount, bytecount, it = generatev1(repo)
                      repo.ui.status(
                          _(b'writing %d bytes for %d files\n') % (bytecount, filecount)
                      )
                      yield struct.pack(b'>QQ', filecount, bytecount)
                      yield struct.pack(b'>H', len(requires) + 1)
                      yield requires + b'\0'
                      # This is where we'll add compression in the future.
                      assert compression == b'UN'
                      progress = repo.ui.makeprogress(
                          _(b'bundle'), total=bytecount, unit=_(b'bytes')
                      )
                      progress.update(0)
                      for chunk in it:
                          progress.increment(step=len(chunk))
                          yield chunk
                      progress.complete()
                  return requirements, gen()
              def consumev1(repo, fp, filecount, bytecount):
                  """Apply the contents from version 1 of a streaming clone file handle.
                  This takes the output from "stream_out" and applies it to the specified
                  repository.
                  Like "stream_out," the status line added by the wire protocol is not
                  handled by this function.
                  """
                  with repo.lock():
                      repo.ui.status(
                          _(b'%d files to transfer, %s of data\n')
                          % (filecount, util.bytecount(bytecount))
                      )
                      progress = repo.ui.makeprogress(
                          _(b'clone'), total=bytecount, unit=_(b'bytes')
                      )
                      progress.update(0)
                      start = util.timer()
                      # TODO: get rid of (potential) inconsistency
                      #
                      # If transaction is started and any @filecache property is
                      # changed at this point, it causes inconsistency between
                      # in-memory cached property and streamclone-ed file on the
                      # disk. Nested transaction prevents transaction scope "clone"
                      # below from writing in-memory changes out at the end of it,
                      # even though in-memory changes are discarded at the end of it
                      # regardless of transaction nesting.
                      #
                      # But transaction nesting can't be simply prohibited, because
                      # nesting occurs also in ordinary case (e.g. enabling
                      # clonebundles).
                      with repo.transaction(b'clone'):
                          with repo.svfs.backgroundclosing(repo.ui, expectedcount=filecount):
                              for i in range(filecount):
                                  # XXX doesn't support '\n' or '\r' in filenames
                                  l = fp.readline()
                                  try:
                                      name, size = l.split(b'\0', 1)
                                      size = int(size)
                                  except (ValueError, TypeError):
                                      raise error.ResponseError(
                                          _(b'unexpected response from remote server:'), l
                                      )
                                  if repo.ui.debugflag:
                                      repo.ui.debug(
                                          b'adding %s (%s)\n' % (name, util.bytecount(size))
                                      )
                                  # for backwards compat, name was partially encoded
                                  path = store.decodedir(name)
                                  with repo.svfs(path, b'w', backgroundclose=True) as ofp:
                                      for chunk in util.filechunkiter(fp, limit=size):
                                          progress.increment(step=len(chunk))
                                          ofp.write(chunk)
                          # force @filecache properties to be reloaded from
                          # streamclone-ed file at next access
                          repo.invalidate(clearfilecache=True)
                      elapsed = util.timer() - start
                      if elapsed <= 0:
                          elapsed = 0.001
                      progress.complete()
                      repo.ui.status(
                          _(b'transferred %s in %.1f seconds (%s/sec)\n')
                          % (
                              util.bytecount(bytecount),
                              elapsed,
                              util.bytecount(bytecount / elapsed),
                          )
                      )
              def readbundle1header(fp):
                  compression = fp.read(2)
                  if compression != b'UN':
                      raise error.Abort(
                          _(
                              b'only uncompressed stream clone bundles are '
                              b'supported; got %s'
                          )
                          % compression
                      )
                  filecount, bytecount = struct.unpack(b'>QQ', fp.read(16))
                  requireslen = struct.unpack(b'>H', fp.read(2))[0]
                  requires = fp.read(requireslen)
                  if not requires.endswith(b'\0'):
                      raise error.Abort(
                          _(
                              b'malformed stream clone bundle: '
                              b'requirements not properly encoded'
                          )
                      )
                  requirements = set(requires.rstrip(b'\0').split(b','))
                  return filecount, bytecount, requirements
              def applybundlev1(repo, fp):
                  """Apply the content from a stream clone bundle version 1.
                  We assume the 4 byte header has been read and validated and the file handle
                  is at the 2 byte compression identifier.
                  """
                  if len(repo):
                      raise error.Abort(
                          _(b'cannot apply stream clone bundle on non-empty repo')
                      )
                  filecount, bytecount, requirements = readbundle1header(fp)
                  missingreqs = requirements - repo.supported
                  if missingreqs:
                      raise error.Abort(
                          _(b'unable to apply stream clone: unsupported format: %s')
                          % b', '.join(sorted(missingreqs))
                      )
                  consumev1(repo, fp, filecount, bytecount)
                  nodemap.post_stream_cleanup(repo)
              class streamcloneapplier:
                  """Class to manage applying streaming clone bundles.
                  We need to wrap ``applybundlev1()`` in a dedicated type to enable bundle
                  readers to perform bundle type-specific functionality.
                  """
                  def __init__(self, fh):
                      self._fh = fh
                  def apply(self, repo):
                      return applybundlev1(repo, self._fh)
              # type of file to stream
              _fileappend = 0  # append only file
              _filefull = 1  # full snapshot file
              # Source of the file
              _srcstore = b's'  # store (svfs)
              _srccache = b'c'  # cache (cache)
              # This is it's own function so extensions can override it.
              def _walkstreamfullstorefiles(repo):
                  """list snapshot file from the store"""
                  fnames = []
                  if not repo.publishing():
                      fnames.append(b'phaseroots')
                  return fnames
              def _filterfull(entry, copy, vfsmap):
                  """actually copy the snapshot files"""
                  src, name, ftype, data = entry
                  if ftype != _filefull:
                      return entry
                  return (src, name, ftype, copy(vfsmap[src].join(name)))
              @contextlib.contextmanager
              def maketempcopies():
                  """return a function to temporary copy file"""
                  files = []
                  dst_dir = pycompat.mkdtemp(prefix=b'hg-clone-')
                  try:
                      def copy(src):
                          fd, dst = pycompat.mkstemp(
                              prefix=os.path.basename(src), dir=dst_dir
                          )
                          os.close(fd)
                          files.append(dst)
                          util.copyfiles(src, dst, hardlink=True)
                          return dst
                      yield copy
                  finally:
                      for tmp in files:
                          util.tryunlink(tmp)
                      util.tryrmdir(dst_dir)
              def _makemap(repo):
                  """make a (src -> vfs) map for the repo"""
                  vfsmap = {
                      _srcstore: repo.svfs,
                      _srccache: repo.cachevfs,
                  }
                  # we keep repo.vfs out of the on purpose, ther are too many danger there
                  # (eg: .hg/hgrc)
                  assert repo.vfs not in vfsmap.values()
                  return vfsmap
              def _emit2(repo, entries, totalfilesize):
                  """actually emit the stream bundle"""
                  vfsmap = _makemap(repo)
                  # we keep repo.vfs out of the on purpose, ther are too many danger there
                  # (eg: .hg/hgrc),
                  #
                  # this assert is duplicated (from _makemap) as author might think this is
                  # fine, while this is really not fine.
                  if repo.vfs in vfsmap.values():
                      raise error.ProgrammingError(
                          b'repo.vfs must not be added to vfsmap for security reasons'
                      )
                  progress = repo.ui.makeprogress(
                      _(b'bundle'), total=totalfilesize, unit=_(b'bytes')
                  )
                  progress.update(0)
                  with maketempcopies() as copy, progress:
                      # copy is delayed until we are in the try
                      entries = [_filterfull(e, copy, vfsmap) for e in entries]
                      yield None  # this release the lock on the repository
                      totalbytecount = 0
                      for src, name, ftype, data in entries:
                          vfs = vfsmap[src]
                          yield src
                          yield util.uvarintencode(len(name))
                          if ftype == _fileappend:
                              fp = vfs(name)
                              size = data
                          elif ftype == _filefull:
                              fp = open(data, b'rb')
                              size = util.fstat(fp).st_size
                          bytecount = 0
                          try:
                              yield util.uvarintencode(size)
                              yield name
                              if size <= 65536:
                                  chunks = (fp.read(size),)
                              else:
                                  chunks = util.filechunkiter(fp, limit=size)
                              for chunk in chunks:
                                  bytecount += len(chunk)
                                  totalbytecount += len(chunk)
                                  progress.update(totalbytecount)
                                  yield chunk
                              if bytecount != size:
                                  # Would most likely be caused by a race due to `hg strip` or
                                  # a revlog split
                                  raise error.Abort(
                                      _(
                                          b'clone could only read %d bytes from %s, but '
                                          b'expected %d bytes'
                                      )
                                      % (bytecount, name, size)
                                  )
                          finally:
                              fp.close()
              def _test_sync_point_walk_1(repo):
                  """a function for synchronisation during tests"""
              def _test_sync_point_walk_2(repo):
                  """a function for synchronisation during tests"""
              def _v2_walk(repo, includes, excludes, includeobsmarkers):
                  """emit a seris of files information useful to clone a repo
                  return (entries, totalfilesize)
                  entries is a list of tuple (vfs-key, file-path, file-type, size)
                  - `vfs-key`: is a key to the right vfs to write the file (see _makemap)
                  - `name`: file path of the file to copy (to be feed to the vfss)
                  - `file-type`: do this file need to be copied with the source lock ?
                  - `size`: the size of the file (or None)
                  """
                  assert repo._currentlock(repo._lockref) is not None
                  entries = []
                  totalfilesize = 0
                  matcher = None
                  if includes or excludes:
                      matcher = narrowspec.match(repo.root, includes, excludes)
                  for entry in _walkstreamfiles(repo, matcher):
-                     if entry.file_size:
+                     for f in entry.files():
+                         if f.file_size:
                          ft = _fileappend
-                         if entry.is_volatile:
+                             if f.is_volatile:
                              ft = _filefull
-                         entries.append(
-                             (_srcstore, entry.unencoded_path, ft, entry.file_size)
+                         )
-                         totalfilesize += entry.file_size
+                             entries.append((_srcstore, f.unencoded_path, ft, f.file_size))
+                             totalfilesize += f.file_size
                  for name in _walkstreamfullstorefiles(repo):
                      if repo.svfs.exists(name):
                          totalfilesize += repo.svfs.lstat(name).st_size
                          entries.append((_srcstore, name, _filefull, None))
                  if includeobsmarkers and repo.svfs.exists(b'obsstore'):
                      totalfilesize += repo.svfs.lstat(b'obsstore').st_size
                      entries.append((_srcstore, b'obsstore', _filefull, None))
                  for name in cacheutil.cachetocopy(repo):
                      if repo.cachevfs.exists(name):
                          totalfilesize += repo.cachevfs.lstat(name).st_size
                          entries.append((_srccache, name, _filefull, None))
                  return entries, totalfilesize
              def generatev2(repo, includes, excludes, includeobsmarkers):
                  """Emit content for version 2 of a streaming clone.
                  the data stream consists the following entries:
 ) A char representing the file destination (eg: store or cache)
 ) A varint containing the length of the filename
 ) A varint containing the length of file data
 ) N bytes containing the filename (the internal, store-agnostic form)
 ) N bytes containing the file data
                  Returns a 3-tuple of (file count, file size, data iterator).
                  """
                  with repo.lock():
                      repo.ui.debug(b'scanning\n')
                      entries, totalfilesize = _v2_walk(
                          repo,
                          includes=includes,
                          excludes=excludes,
                          includeobsmarkers=includeobsmarkers,
                      )
                      chunks = _emit2(repo, entries, totalfilesize)
                      first = next(chunks)
                      assert first is None
                      _test_sync_point_walk_1(repo)
                  _test_sync_point_walk_2(repo)
                  return len(entries), totalfilesize, chunks
              @contextlib.contextmanager
              def nested(*ctxs):
                  this = ctxs[0]
                  rest = ctxs[1:]
                  with this:
                      if rest:
                          with nested(*rest):
                              yield
                      else:
                          yield
              def consumev2(repo, fp, filecount, filesize):
                  """Apply the contents from a version 2 streaming clone.
                  Data is read from an object that only needs to provide a ``read(size)``
                  method.
                  """
                  with repo.lock():
                      repo.ui.status(
                          _(b'%d files to transfer, %s of data\n')
                          % (filecount, util.bytecount(filesize))
                      )
                      start = util.timer()
                      progress = repo.ui.makeprogress(
                          _(b'clone'), total=filesize, unit=_(b'bytes')
                      )
                      progress.update(0)
                      vfsmap = _makemap(repo)
                      # we keep repo.vfs out of the on purpose, ther are too many danger
                      # there (eg: .hg/hgrc),
                      #
                      # this assert is duplicated (from _makemap) as author might think this
                      # is fine, while this is really not fine.
                      if repo.vfs in vfsmap.values():
                          raise error.ProgrammingError(
                              b'repo.vfs must not be added to vfsmap for security reasons'
                          )
                      with repo.transaction(b'clone'):
                          ctxs = (vfs.backgroundclosing(repo.ui) for vfs in vfsmap.values())
                          with nested(*ctxs):
                              for i in range(filecount):
                                  src = util.readexactly(fp, 1)
                                  vfs = vfsmap[src]
                                  namelen = util.uvarintdecodestream(fp)
                                  datalen = util.uvarintdecodestream(fp)
                                  name = util.readexactly(fp, namelen)
                                  if repo.ui.debugflag:
                                      repo.ui.debug(
                                          b'adding [%s] %s (%s)\n'
                                          % (src, name, util.bytecount(datalen))
                                      )
                                  with vfs(name, b'w') as ofp:
                                      for chunk in util.filechunkiter(fp, limit=datalen):
                                          progress.increment(step=len(chunk))
                                          ofp.write(chunk)
                          # force @filecache properties to be reloaded from
                          # streamclone-ed file at next access
                          repo.invalidate(clearfilecache=True)
                      elapsed = util.timer() - start
                      if elapsed <= 0:
                          elapsed = 0.001
                      repo.ui.status(
                          _(b'transferred %s in %.1f seconds (%s/sec)\n')
                          % (
                              util.bytecount(progress.pos),
                              elapsed,
                              util.bytecount(progress.pos / elapsed),
                          )
                      )
                      progress.complete()
              def applybundlev2(repo, fp, filecount, filesize, requirements):
                  from . import localrepo
                  missingreqs = [r for r in requirements if r not in repo.supported]
                  if missingreqs:
                      raise error.Abort(
                          _(b'unable to apply stream clone: unsupported format: %s')
                          % b', '.join(sorted(missingreqs))
                      )
                  consumev2(repo, fp, filecount, filesize)
                  repo.requirements = new_stream_clone_requirements(
                      repo.requirements,
                      requirements,
                  )
                  repo.svfs.options = localrepo.resolvestorevfsoptions(
                      repo.ui, repo.requirements, repo.features
                  )
                  scmutil.writereporequirements(repo)
                  nodemap.post_stream_cleanup(repo)
              def _copy_files(src_vfs_map, dst_vfs_map, entries, progress):
                  hardlink = [True]
                  def copy_used():
                      hardlink[0] = False
                      progress.topic = _(b'copying')
                  for k, path, size in entries:
                      src_vfs = src_vfs_map[k]
                      dst_vfs = dst_vfs_map[k]
                      src_path = src_vfs.join(path)
                      dst_path = dst_vfs.join(path)
                      # We cannot use dirname and makedirs of dst_vfs here because the store
                      # encoding confuses them. See issue 6581 for details.
                      dirname = os.path.dirname(dst_path)
                      if not os.path.exists(dirname):
                          util.makedirs(dirname)
                      dst_vfs.register_file(path)
                      # XXX we could use the #nb_bytes argument.
                      util.copyfile(
                          src_path,
                          dst_path,
                          hardlink=hardlink[0],
                          no_hardlink_cb=copy_used,
                          check_fs_hardlink=False,
                      )
                      progress.increment()
                  return hardlink[0]
              def local_copy(src_repo, dest_repo):
                  """copy all content from one local repository to another
                  This is useful for local clone"""
                  src_store_requirements = {
                      r
                      for r in src_repo.requirements
                      if r not in requirementsmod.WORKING_DIR_REQUIREMENTS
                  }
                  dest_store_requirements = {
                      r
                      for r in dest_repo.requirements
                      if r not in requirementsmod.WORKING_DIR_REQUIREMENTS
                  }
                  assert src_store_requirements == dest_store_requirements
                  with dest_repo.lock():
                      with src_repo.lock():
                          # bookmark is not integrated to the streaming as it might use the
                          # `repo.vfs` and they are too many sentitive data accessible
                          # through `repo.vfs` to expose it to streaming clone.
                          src_book_vfs = bookmarks.bookmarksvfs(src_repo)
                          srcbookmarks = src_book_vfs.join(b'bookmarks')
                          bm_count = 0
                          if os.path.exists(srcbookmarks):
                              bm_count = 1
                          entries, totalfilesize = _v2_walk(
                              src_repo,
                              includes=None,
                              excludes=None,
                              includeobsmarkers=True,
                          )
                          src_vfs_map = _makemap(src_repo)
                          dest_vfs_map = _makemap(dest_repo)
                          progress = src_repo.ui.makeprogress(
                              topic=_(b'linking'),
                              total=len(entries) + bm_count,
                              unit=_(b'files'),
                          )
                          # copy  files
                          #
                          # We could copy the full file while the source repository is locked
                          # and the other one without the lock. However, in the linking case,
                          # this would also requires checks that nobody is appending any data
                          # to the files while we do the clone, so this is not done yet. We
                          # could do this blindly when copying files.
                          files = ((k, path, size) for k, path, ftype, size in entries)
                          hardlink = _copy_files(src_vfs_map, dest_vfs_map, files, progress)
                          # copy bookmarks over
                          if bm_count:
                              dst_book_vfs = bookmarks.bookmarksvfs(dest_repo)
                              dstbookmarks = dst_book_vfs.join(b'bookmarks')
                              util.copyfile(srcbookmarks, dstbookmarks)
                      progress.complete()
                      if hardlink:
                          msg = b'linked %d files\n'
                      else:
                          msg = b'copied %d files\n'
                      src_repo.ui.debug(msg % (len(entries) + bm_count))
                      with dest_repo.transaction(b"localclone") as tr:
                          dest_repo.store.write(tr)
                      # clean up transaction file as they do not make sense
                      transaction.cleanup_undo_files(dest_repo.ui.warn, dest_repo.vfs_map)

mercurial/verify.py

0 +6 -4

              # verify.py - repository integrity checking for Mercurial
              #
              # Copyright 2006, 2007 Olivia Mackall <olivia@selenic.com>
              #
              # This software may be used and distributed according to the terms of the
              # GNU General Public License version 2 or any later version.
              import os
              from .i18n import _
              from .node import short
              from .utils import stringutil
              from . import (
                  error,
                  pycompat,
                  requirements,
                  revlog,
                  util,
              )
              VERIFY_DEFAULT = 0
              VERIFY_FULL = 1
              def verify(repo, level=None):
                  with repo.lock():
                      v = verifier(repo, level)
                      return v.verify()
              def _normpath(f):
                  # under hg < 2.4, convert didn't sanitize paths properly, so a
                  # converted repo may contain repeated slashes
                  while b'//' in f:
                      f = f.replace(b'//', b'/')
                  return f
              HINT_FNCACHE = _(
                  b'hint: run "hg debugrebuildfncache" to recover from corrupt fncache\n'
              )
              WARN_PARENT_DIR_UNKNOWN_REV = _(
                  b"parent-directory manifest refers to unknown revision %s"
              )
              WARN_UNKNOWN_COPY_SOURCE = _(
                  b"warning: copy source of '%s' not in parents of %s"
              )
              WARN_NULLID_COPY_SOURCE = _(
                  b"warning: %s@%s: copy source revision is nullid %s:%s\n"
              )
              class verifier:
                  def __init__(self, repo, level=None):
                      self.repo = repo.unfiltered()
                      self.ui = repo.ui
                      self.match = repo.narrowmatch()
                      if level is None:
                          level = VERIFY_DEFAULT
                      self._level = level
                      self.badrevs = set()
                      self.errors = 0
                      self.warnings = 0
                      self.havecl = len(repo.changelog) > 0
                      self.havemf = len(repo.manifestlog.getstorage(b'')) > 0
                      self.revlogv1 = repo.changelog._format_version != revlog.REVLOGV0
                      self.lrugetctx = util.lrucachefunc(repo.unfiltered().__getitem__)
                      self.refersmf = False
                      self.fncachewarned = False
                      # developer config: verify.skipflags
                      self.skipflags = repo.ui.configint(b'verify', b'skipflags')
                      self.warnorphanstorefiles = True
                  def _warn(self, msg):
                      """record a "warning" level issue"""
                      self.ui.warn(msg + b"\n")
                      self.warnings += 1
                  def _err(self, linkrev, msg, filename=None):
                      """record a "error" level issue"""
                      if linkrev is not None:
                          self.badrevs.add(linkrev)
                          linkrev = b"%d" % linkrev
                      else:
                          linkrev = b'?'
                      msg = b"%s: %s" % (linkrev, msg)
                      if filename:
                          msg = b"%s@%s" % (filename, msg)
                      self.ui.warn(b" " + msg + b"\n")
                      self.errors += 1
                  def _exc(self, linkrev, msg, inst, filename=None):
                      """record exception raised during the verify process"""
                      fmsg = stringutil.forcebytestr(inst)
                      if not fmsg:
                          fmsg = pycompat.byterepr(inst)
                      self._err(linkrev, b"%s: %s" % (msg, fmsg), filename)
                  def _checkrevlog(self, obj, name, linkrev):
                      """verify high level property of a revlog
                      - revlog is present,
                      - revlog is non-empty,
                      - sizes (index and data) are correct,
                      - revlog's format version is correct.
                      """
                      if not len(obj) and (self.havecl or self.havemf):
                          self._err(linkrev, _(b"empty or missing %s") % name)
                          return
                      d = obj.checksize()
                      if d[0]:
                          self._err(None, _(b"data length off by %d bytes") % d[0], name)
                      if d[1]:
                          self._err(None, _(b"index contains %d extra bytes") % d[1], name)
                      if obj._format_version != revlog.REVLOGV0:
                          if not self.revlogv1:
                              self._warn(_(b"warning: `%s' uses revlog format 1") % name)
                      elif self.revlogv1:
                          self._warn(_(b"warning: `%s' uses revlog format 0") % name)
                  def _checkentry(self, obj, i, node, seen, linkrevs, f):
                      """verify a single revlog entry
                      arguments are:
                      - obj:      the source revlog
                      - i:        the revision number
                      - node:     the revision node id
                      - seen:     nodes previously seen for this revlog
                      - linkrevs: [changelog-revisions] introducing "node"
                      - f:        string label ("changelog", "manifest", or filename)
                      Performs the following checks:
                      - linkrev points to an existing changelog revision,
                      - linkrev points to a changelog revision that introduces this revision,
                      - linkrev points to the lowest of these changesets,
                      - both parents exist in the revlog,
                      - the revision is not duplicated.
                      Return the linkrev of the revision (or None for changelog's revisions).
                      """
                      lr = obj.linkrev(obj.rev(node))
                      if lr < 0 or (self.havecl and lr not in linkrevs):
                          if lr < 0 or lr >= len(self.repo.changelog):
                              msg = _(b"rev %d points to nonexistent changeset %d")
                          else:
                              msg = _(b"rev %d points to unexpected changeset %d")
                          self._err(None, msg % (i, lr), f)
                          if linkrevs:
                              if f and len(linkrevs) > 1:
                                  try:
                                      # attempt to filter down to real linkrevs
                                      linkrevs = []
                                      for lr in linkrevs:
                                          if self.lrugetctx(lr)[f].filenode() == node:
                                              linkrevs.append(lr)
                                  except Exception:
                                      pass
                              msg = _(b" (expected %s)")
                              msg %= b" ".join(map(pycompat.bytestr, linkrevs))
                              self._warn(msg)
                          lr = None  # can't be trusted
                      try:
                          p1, p2 = obj.parents(node)
                          if p1 not in seen and p1 != self.repo.nullid:
                              msg = _(b"unknown parent 1 %s of %s") % (short(p1), short(node))
                              self._err(lr, msg, f)
                          if p2 not in seen and p2 != self.repo.nullid:
                              msg = _(b"unknown parent 2 %s of %s") % (short(p2), short(node))
                              self._err(lr, msg, f)
                      except Exception as inst:
                          self._exc(lr, _(b"checking parents of %s") % short(node), inst, f)
                      if node in seen:
                          self._err(lr, _(b"duplicate revision %d (%d)") % (i, seen[node]), f)
                      seen[node] = i
                      return lr
                  def verify(self):
                      """verify the content of the Mercurial repository
                      This method run all verifications, displaying issues as they are found.
                      return 1 if any error have been encountered, 0 otherwise."""
                      # initial validation and generic report
                      repo = self.repo
                      ui = repo.ui
                      if not repo.url().startswith(b'file:'):
                          raise error.Abort(_(b"cannot verify bundle or remote repos"))
                      if os.path.exists(repo.sjoin(b"journal")):
                          ui.warn(_(b"abandoned transaction found - run hg recover\n"))
                      if ui.verbose or not self.revlogv1:
                          ui.status(
                              _(b"repository uses revlog format %d\n")
                              % (self.revlogv1 and 1 or 0)
                          )
                      # data verification
                      mflinkrevs, filelinkrevs = self._verifychangelog()
                      filenodes = self._verifymanifest(mflinkrevs)
                      del mflinkrevs
                      self._crosscheckfiles(filelinkrevs, filenodes)
                      totalfiles, filerevisions = self._verifyfiles(filenodes, filelinkrevs)
                      if self.errors:
                          ui.warn(_(b"not checking dirstate because of previous errors\n"))
                          dirstate_errors = 0
                      else:
                          dirstate_errors = self._verify_dirstate()
                      # final report
                      ui.status(
                          _(b"checked %d changesets with %d changes to %d files\n")
                          % (len(repo.changelog), filerevisions, totalfiles)
                      )
                      if self.warnings:
                          ui.warn(_(b"%d warnings encountered!\n") % self.warnings)
                      if self.fncachewarned:
                          ui.warn(HINT_FNCACHE)
                      if self.errors:
                          ui.warn(_(b"%d integrity errors encountered!\n") % self.errors)
                          if self.badrevs:
                              msg = _(b"(first damaged changeset appears to be %d)\n")
                              msg %= min(self.badrevs)
                              ui.warn(msg)
                          if dirstate_errors:
                              ui.warn(
                                  _(b"dirstate inconsistent with current parent's manifest\n")
                              )
                              ui.warn(_(b"%d dirstate errors\n") % dirstate_errors)
                          return 1
                      return 0
                  def _verifychangelog(self):
                      """verify the changelog of a repository
                      The following checks are performed:
                      - all of `_checkrevlog` checks,
                      - all of `_checkentry` checks (for each revisions),
                      - each revision can be read.
                      The function returns some of the data observed in the changesets as a
                      (mflinkrevs, filelinkrevs) tuples:
                      - mflinkrevs:   is a { manifest-node -> [changelog-rev] } mapping
                      - filelinkrevs: is a { file-path -> [changelog-rev] } mapping
                      If a matcher was specified, filelinkrevs will only contains matched
                      files.
                      """
                      ui = self.ui
                      repo = self.repo
                      match = self.match
                      cl = repo.changelog
                      ui.status(_(b"checking changesets\n"))
                      mflinkrevs = {}
                      filelinkrevs = {}
                      seen = {}
                      self._checkrevlog(cl, b"changelog", 0)
                      progress = ui.makeprogress(
                          _(b'checking'), unit=_(b'changesets'), total=len(repo)
                      )
                      for i in repo:
                          progress.update(i)
                          n = cl.node(i)
                          self._checkentry(cl, i, n, seen, [i], b"changelog")
                          try:
                              changes = cl.read(n)
                              if changes[0] != self.repo.nullid:
                                  mflinkrevs.setdefault(changes[0], []).append(i)
                                  self.refersmf = True
                              for f in changes[3]:
                                  if match(f):
                                      filelinkrevs.setdefault(_normpath(f), []).append(i)
                          except Exception as inst:
                              self.refersmf = True
                              self._exc(i, _(b"unpacking changeset %s") % short(n), inst)
                      progress.complete()
                      return mflinkrevs, filelinkrevs
                  def _verifymanifest(
                      self, mflinkrevs, dir=b"", storefiles=None, subdirprogress=None
                  ):
                      """verify the manifestlog content
                      Inputs:
                      - mflinkrevs:     a {manifest-node -> [changelog-revisions]} mapping
                      - dir:            a subdirectory to check (for tree manifest repo)
                      - storefiles:     set of currently "orphan" files.
                      - subdirprogress: a progress object
                      This function checks:
                      * all of `_checkrevlog` checks (for all manifest related revlogs)
                      * all of `_checkentry` checks (for all manifest related revisions)
                      * nodes for subdirectory exists in the sub-directory manifest
                      * each manifest entries have a file path
                      * each manifest node refered in mflinkrevs exist in the manifest log
                      If tree manifest is in use and a matchers is specified, only the
                      sub-directories matching it will be verified.
                      return a two level mapping:
                          {"path" -> { filenode -> changelog-revision}}
                      This mapping primarily contains entries for every files in the
                      repository. In addition, when tree-manifest is used, it also contains
                      sub-directory entries.
                      If a matcher is provided, only matching paths will be included.
                      """
                      repo = self.repo
                      ui = self.ui
                      match = self.match
                      mfl = self.repo.manifestlog
                      mf = mfl.getstorage(dir)
                      if not dir:
                          self.ui.status(_(b"checking manifests\n"))
                      filenodes = {}
                      subdirnodes = {}
                      seen = {}
                      label = b"manifest"
                      if dir:
                          label = dir
                          revlogfiles = mf.files()
                          storefiles.difference_update(revlogfiles)
                          if subdirprogress:  # should be true since we're in a subdirectory
                              subdirprogress.increment()
                      if self.refersmf:
                          # Do not check manifest if there are only changelog entries with
                          # null manifests.
                          self._checkrevlog(mf._revlog, label, 0)
                      progress = ui.makeprogress(
                          _(b'checking'), unit=_(b'manifests'), total=len(mf)
                      )
                      for i in mf:
                          if not dir:
                              progress.update(i)
                          n = mf.node(i)
                          lr = self._checkentry(mf, i, n, seen, mflinkrevs.get(n, []), label)
                          if n in mflinkrevs:
                              del mflinkrevs[n]
                          elif dir:
                              msg = _(b"%s not in parent-directory manifest") % short(n)
                              self._err(lr, msg, label)
                          else:
                              self._err(lr, _(b"%s not in changesets") % short(n), label)
                          try:
                              mfdelta = mfl.get(dir, n).readdelta(shallow=True)
                              for f, fn, fl in mfdelta.iterentries():
                                  if not f:
                                      self._err(lr, _(b"entry without name in manifest"))
                                  elif f == b"/dev/null":  # ignore this in very old repos
                                      continue
                                  fullpath = dir + _normpath(f)
                                  if fl == b't':
                                      if not match.visitdir(fullpath):
                                          continue
                                      sdn = subdirnodes.setdefault(fullpath + b'/', {})
                                      sdn.setdefault(fn, []).append(lr)
                                  else:
                                      if not match(fullpath):
                                          continue
                                      filenodes.setdefault(fullpath, {}).setdefault(fn, lr)
                          except Exception as inst:
                              self._exc(lr, _(b"reading delta %s") % short(n), inst, label)
                          if self._level >= VERIFY_FULL:
                              try:
                                  # Various issues can affect manifest. So we read each full
                                  # text from storage. This triggers the checks from the core
                                  # code (eg: hash verification, filename are ordered, etc.)
                                  mfdelta = mfl.get(dir, n).read()
                              except Exception as inst:
                                  msg = _(b"reading full manifest %s") % short(n)
                                  self._exc(lr, msg, inst, label)
                      if not dir:
                          progress.complete()
                      if self.havemf:
                          # since we delete entry in `mflinkrevs` during iteration, any
                          # remaining entries are "missing". We need to issue errors for them.
                          changesetpairs = [(c, m) for m in mflinkrevs for c in mflinkrevs[m]]
                          for c, m in sorted(changesetpairs):
                              if dir:
                                  self._err(c, WARN_PARENT_DIR_UNKNOWN_REV % short(m), label)
                              else:
                                  msg = _(b"changeset refers to unknown revision %s")
                                  msg %= short(m)
                                  self._err(c, msg, label)
                      if not dir and subdirnodes:
                          self.ui.status(_(b"checking directory manifests\n"))
                          storefiles = set()
                          subdirs = set()
                          revlogv1 = self.revlogv1
                          undecodable = []
                          for entry in repo.store.datafiles(undecodable=undecodable):
-                             f = entry.unencoded_path
-                             size = entry.file_size
+                             for file_ in entry.files():
+                                 f = file_.unencoded_path
+                                 size = file_.file_size
                              if (size > 0 or not revlogv1) and f.startswith(b'meta/'):
                                  storefiles.add(_normpath(f))
                                  subdirs.add(os.path.dirname(f))
                          for f in undecodable:
                              self._err(None, _(b"cannot decode filename '%s'") % f)
                          subdirprogress = ui.makeprogress(
                              _(b'checking'), unit=_(b'manifests'), total=len(subdirs)
                          )
                      for subdir, linkrevs in subdirnodes.items():
                          subdirfilenodes = self._verifymanifest(
                              linkrevs, subdir, storefiles, subdirprogress
                          )
                          for f, onefilenodes in subdirfilenodes.items():
                              filenodes.setdefault(f, {}).update(onefilenodes)
                      if not dir and subdirnodes:
                          assert subdirprogress is not None  # help pytype
                          subdirprogress.complete()
                          if self.warnorphanstorefiles:
                              for f in sorted(storefiles):
                                  self._warn(_(b"warning: orphan data file '%s'") % f)
                      return filenodes
                  def _crosscheckfiles(self, filelinkrevs, filenodes):
                      repo = self.repo
                      ui = self.ui
                      ui.status(_(b"crosschecking files in changesets and manifests\n"))
                      total = len(filelinkrevs) + len(filenodes)
                      progress = ui.makeprogress(
                          _(b'crosschecking'), unit=_(b'files'), total=total
                      )
                      if self.havemf:
                          for f in sorted(filelinkrevs):
                              progress.increment()
                              if f not in filenodes:
                                  lr = filelinkrevs[f][0]
                                  self._err(lr, _(b"in changeset but not in manifest"), f)
                      if self.havecl:
                          for f in sorted(filenodes):
                              progress.increment()
                              if f not in filelinkrevs:
                                  try:
                                      fl = repo.file(f)
                                      lr = min([fl.linkrev(fl.rev(n)) for n in filenodes[f]])
                                  except Exception:
                                      lr = None
                                  self._err(lr, _(b"in manifest but not in changeset"), f)
                      progress.complete()
                  def _verifyfiles(self, filenodes, filelinkrevs):
                      repo = self.repo
                      ui = self.ui
                      lrugetctx = self.lrugetctx
                      revlogv1 = self.revlogv1
                      havemf = self.havemf
                      ui.status(_(b"checking files\n"))
                      storefiles = set()
                      undecodable = []
                      for entry in repo.store.datafiles(undecodable=undecodable):
-                         size = entry.file_size
-                         f = entry.unencoded_path
+                         for file_ in entry.files():
+                             size = file_.file_size
+                             f = file_.unencoded_path
                          if (size > 0 or not revlogv1) and f.startswith(b'data/'):
                              storefiles.add(_normpath(f))
                      for f in undecodable:
                          self._err(None, _(b"cannot decode filename '%s'") % f)
                      state = {
                          # TODO this assumes revlog storage for changelog.
                          b'expectedversion': self.repo.changelog._format_version,
                          b'skipflags': self.skipflags,
                          # experimental config: censor.policy
                          b'erroroncensored': ui.config(b'censor', b'policy') == b'abort',
                      }
                      files = sorted(set(filenodes) | set(filelinkrevs))
                      revisions = 0
                      progress = ui.makeprogress(
                          _(b'checking'), unit=_(b'files'), total=len(files)
                      )
                      for i, f in enumerate(files):
                          progress.update(i, item=f)
                          try:
                              linkrevs = filelinkrevs[f]
                          except KeyError:
                              # in manifest but not in changelog
                              linkrevs = []
                          if linkrevs:
                              lr = linkrevs[0]
                          else:
                              lr = None
                          try:
                              fl = repo.file(f)
                          except error.StorageError as e:
                              self._err(lr, _(b"broken revlog! (%s)") % e, f)
                              continue
                          for ff in fl.files():
                              try:
                                  storefiles.remove(ff)
                              except KeyError:
                                  if self.warnorphanstorefiles:
                                      msg = _(b" warning: revlog '%s' not in fncache!")
                                      self._warn(msg % ff)
                                      self.fncachewarned = True
                          if not len(fl) and (self.havecl or self.havemf):
                              self._err(lr, _(b"empty or missing %s") % f)
                          else:
                              # Guard against implementations not setting this.
                              state[b'skipread'] = set()
                              state[b'safe_renamed'] = set()
                              for problem in fl.verifyintegrity(state):
                                  if problem.node is not None:
                                      linkrev = fl.linkrev(fl.rev(problem.node))
                                  else:
                                      linkrev = None
                                  if problem.warning:
                                      self._warn(problem.warning)
                                  elif problem.error:
                                      linkrev_msg = linkrev if linkrev is not None else lr
                                      self._err(linkrev_msg, problem.error, f)
                                  else:
                                      raise error.ProgrammingError(
                                          b'problem instance does not set warning or error '
                                          b'attribute: %s' % problem.msg
                                      )
                          seen = {}
                          for i in fl:
                              revisions += 1
                              n = fl.node(i)
                              lr = self._checkentry(fl, i, n, seen, linkrevs, f)
                              if f in filenodes:
                                  if havemf and n not in filenodes[f]:
                                      self._err(lr, _(b"%s not in manifests") % (short(n)), f)
                                  else:
                                      del filenodes[f][n]
                              if n in state[b'skipread'] and n not in state[b'safe_renamed']:
                                  continue
                              # check renames
                              try:
                                  # This requires resolving fulltext (at least on revlogs,
                                  # though not with LFS revisions). We may want
                                  # ``verifyintegrity()`` to pass a set of nodes with
                                  # rename metadata as an optimization.
                                  rp = fl.renamed(n)
                                  if rp:
                                      if lr is not None and ui.verbose:
                                          ctx = lrugetctx(lr)
                                          if not any(rp[0] in pctx for pctx in ctx.parents()):
                                              self._warn(WARN_UNKNOWN_COPY_SOURCE % (f, ctx))
                                      fl2 = repo.file(rp[0])
                                      if not len(fl2):
                                          m = _(b"empty or missing copy source revlog %s:%s")
                                          self._err(lr, m % (rp[0], short(rp[1])), f)
                                      elif rp[1] == self.repo.nullid:
                                          msg = WARN_NULLID_COPY_SOURCE
                                          msg %= (f, lr, rp[0], short(rp[1]))
                                          ui.note(msg)
                                      else:
                                          fl2.rev(rp[1])
                              except Exception as inst:
                                  self._exc(
                                      lr, _(b"checking rename of %s") % short(n), inst, f
                                  )
                          # cross-check
                          if f in filenodes:
                              fns = [(v, k) for k, v in filenodes[f].items()]
                              for lr, node in sorted(fns):
                                  msg = _(b"manifest refers to unknown revision %s")
                                  self._err(lr, msg % short(node), f)
                      progress.complete()
                      if self.warnorphanstorefiles:
                          for f in sorted(storefiles):
                              self._warn(_(b"warning: orphan data file '%s'") % f)
                      return len(files), revisions
                  def _verify_dirstate(self):
                      """Check that the dirstate is consistent with the parent's manifest"""
                      repo = self.repo
                      ui = self.ui
                      ui.status(_(b"checking dirstate\n"))
                      parent1, parent2 = repo.dirstate.parents()
                      m1 = repo[parent1].manifest()
                      m2 = repo[parent2].manifest()
                      dirstate_errors = 0
                      is_narrow = requirements.NARROW_REQUIREMENT in repo.requirements
                      narrow_matcher = repo.narrowmatch() if is_narrow else None
                      for err in repo.dirstate.verify(m1, m2, parent1, narrow_matcher):
                          ui.error(err)
                          dirstate_errors += 1
                      if dirstate_errors:
                          self.errors += dirstate_errors
                      return dirstate_errors

General Comments 0

Write
Preview

You need to be logged in to leave comments. Login now

No TODOs yet

	Site-wide shortcuts
/	Use quick search box
g h	Goto home page
g g	Goto my private gists page
g G	Goto my public gists page
g 0-9	Goto bookmarked items from 0-9
n r	New repository page
n g	New gist page

	Repositories
g s	Goto summary page
g c	Goto changelog page
g f	Goto files page
g F	Goto files page with file search activated
g p	Goto pull requests page
g o	Goto repository settings
g O	Goto repository access permissions settings
t s	Toggle sidebar on some pages