upstream/mercurial-mirror Files · mercurial/streamclone.py

findrenames: Optimise "addremove -s100" by matching files by their SHA1 hashes....

findrenames: Optimise "addremove -s100" by matching files by their SHA1 hashes. We speed up 'findrenames' for the usecase when a user specifies they want a similarity of 100% by matching files by their exact SHA1 hash value. This reduces the number of comparisons required to find exact matches from O(n^2) to O(n). While it would be nice if we could just use mercurial's pre-calculated SHA1 hash for existing files, this hash includes the file's ancestor information making it unsuitable for our purposes. Instead, we calculate the hash of old content from scratch. The following benchmarks were taken on the current head of crew: addremove 100% similarity: rm -rf *; hg up -C; mv tests tests.new hg --time addremove -s100 --dry-run before: real 176.350 secs (user 128.890+0.000 sys 47.430+0.000) after: real 2.130 secs (user 1.890+0.000 sys 0.240+0.000) addremove 75% similarity: rm -rf *; hg up -C; mv tests tests.new; \ for i in tests.new/*; do echo x >> $i; done hg --time addremove -s75 --dry-run before: real 264.560 secs (user 215.130+0.000 sys 49.410+0.000) after: real 218.710 secs (user 172.790+0.000 sys 45.870+0.000)

Nicolas Dumazet - - Load All Authors

File last commit:

r10905:13a1b2fb default


                r11060:e6df0177

default

Download file

             streamclone.py
        
                    69 lines
            
             | 2.2 KiB
            
                | text/x-python
            
             |
                PythonLexer
            
             / mercurial / streamclone.py
          
                    History
                
                 |
                  Annotation
                 | Raw
                 |Copy content
                 |Copy permalink

      # streamclone.py - streaming clone server support for mercurial

      #

      # Copyright 2006 Vadim Gelfer <vadim.gelfer@gmail.com>

      #

      # This software may be used and distributed according to the terms of the

      # GNU General Public License version 2 or any later version.

      import util, error

      from mercurial import store

      class StreamException(Exception):

          def __init__(self, code):

              Exception.__init__(self)

              self.code = code

          def __str__(self):

              return '%i\n' % self.code

      # if server supports streaming clone, it advertises "stream"

      # capability with value that is version+flags of repo it is serving.

      # client only streams if it can read that repo format.

      # stream file format is simple.

      #

      # server writes out line that says how many files, how many total

      # bytes.  separator is ascii space, byte counts are strings.

      #

      # then for each file:

      #

      #   server writes out line that says filename, how many bytes in

      #   file.  separator is ascii nul, byte count is string.

      #

      #   server writes out raw file data.

      def allowed(ui):

          return ui.configbool('server', 'uncompressed', True, untrusted=True)

      def stream_out(repo):

          '''stream out all metadata files in repository.

          writes to file-like object, must support write() and optional flush().'''

          if not allowed(repo.ui):

              raise StreamException(1)

          entries = []

          total_bytes = 0

          try:

              # get consistent snapshot of repo, lock during scan

              lock = repo.lock()

              try:

                  repo.ui.debug('scanning\n')

                  for name, ename, size in repo.store.walk():

                      entries.append((name, size))

                      total_bytes += size

              finally:

                  lock.release()

          except error.LockError:

              raise StreamException(2)

          yield '0\n'

          repo.ui.debug('%d files, %d bytes to transfer\n' %

                        (len(entries), total_bytes))

          yield '%d %d\n' % (len(entries), total_bytes)

          for name, size in entries:

              repo.ui.debug('sending %s (%d bytes)\n' % (name, size))

              # partially encode name over the wire for backwards compat

              yield '%s\0%d\n' % (store.encodedir(name), size)

              for chunk in util.filechunkiter(repo.sopener(name), limit=size):

                  yield chunk

	Site-wide shortcuts
/	Use quick search box
g h	Goto home page
g g	Goto my private gists page
g G	Goto my public gists page
g 0-9	Goto bookmarked items from 0-9
n r	New repository page
n g	New gist page

	Repositories
g s	Goto summary page
g c	Goto changelog page
g f	Goto files page
g F	Goto files page with file search activated
g p	Goto pull requests page
g o	Goto repository settings
g O	Goto repository access permissions settings
t s	Toggle sidebar on some pages

				# streamclone.py - streaming clone server support for mercurial
				#
				# Copyright 2006 Vadim Gelfer <vadim.gelfer@gmail.com>
				#
				# This software may be used and distributed according to the terms of the
				# GNU General Public License version 2 or any later version.

				import util, error

				from mercurial import store

				class StreamException(Exception):
				def __init__(self, code):
				Exception.__init__(self)
				self.code = code
				def __str__(self):
				return '%i\n' % self.code

				# if server supports streaming clone, it advertises "stream"
				# capability with value that is version+flags of repo it is serving.
				# client only streams if it can read that repo format.

				# stream file format is simple.
				#
				# server writes out line that says how many files, how many total
				# bytes. separator is ascii space, byte counts are strings.
				#
				# then for each file:
				#
				# server writes out line that says filename, how many bytes in
				# file. separator is ascii nul, byte count is string.
				#
				# server writes out raw file data.

				def allowed(ui):
				return ui.configbool('server', 'uncompressed', True, untrusted=True)

				def stream_out(repo):
				'''stream out all metadata files in repository.
				writes to file-like object, must support write() and optional flush().'''

				if not allowed(repo.ui):
				raise StreamException(1)

				entries = []
				total_bytes = 0
				try:
				# get consistent snapshot of repo, lock during scan
				lock = repo.lock()
				try:
				repo.ui.debug('scanning\n')
				for name, ename, size in repo.store.walk():
				entries.append((name, size))
				total_bytes += size
				finally:
				lock.release()
				except error.LockError:
				raise StreamException(2)

				yield '0\n'
				repo.ui.debug('%d files, %d bytes to transfer\n' %
				(len(entries), total_bytes))
				yield '%d %d\n' % (len(entries), total_bytes)
				for name, size in entries:
				repo.ui.debug('sending %s (%d bytes)\n' % (name, size))
				# partially encode name over the wire for backwards compat
				yield '%s\0%d\n' % (store.encodedir(name), size)
				for chunk in util.filechunkiter(repo.sopener(name), limit=size):
				yield chunk