upstream/mercurial-mirror Files · mercurial/filelog.py

findrenames: Optimise "addremove -s100" by matching files by their SHA1 hashes....

findrenames: Optimise "addremove -s100" by matching files by their SHA1 hashes. We speed up 'findrenames' for the usecase when a user specifies they want a similarity of 100% by matching files by their exact SHA1 hash value. This reduces the number of comparisons required to find exact matches from O(n^2) to O(n). While it would be nice if we could just use mercurial's pre-calculated SHA1 hash for existing files, this hash includes the file's ancestor information making it unsuitable for our purposes. Instead, we calculate the hash of old content from scratch. The following benchmarks were taken on the current head of crew: addremove 100% similarity: rm -rf *; hg up -C; mv tests tests.new hg --time addremove -s100 --dry-run before: real 176.350 secs (user 128.890+0.000 sys 47.430+0.000) after: real 2.130 secs (user 1.890+0.000 sys 0.240+0.000) addremove 75% similarity: rm -rf *; hg up -C; mv tests tests.new; \ for i in tests.new/*; do echo x >> $i; done hg --time addremove -s75 --dry-run before: real 264.560 secs (user 215.130+0.000 sys 49.410+0.000) after: real 218.710 secs (user 172.790+0.000 sys 45.870+0.000)

Benoit Boissinot - - Load All Authors

File last commit:

r10706:d8d1b56d default


                r11060:e6df0177

default

Download file

             filelog.py
        
                    66 lines
            
             | 2.0 KiB
            
                | text/x-python
            
             |
                PythonLexer
            
             / mercurial / filelog.py
          
                    History
                
                 |
                  Annotation
                 | Raw
                 |Copy content
                 |Copy permalink

      # filelog.py - file history class for mercurial

      #

      # Copyright 2005-2007 Matt Mackall <mpm@selenic.com>

      #

      # This software may be used and distributed according to the terms of the

      # GNU General Public License version 2 or any later version.

      import revlog

      class filelog(revlog.revlog):

          def __init__(self, opener, path):

              revlog.revlog.__init__(self, opener,

                              "/".join(("data", path + ".i")))

          def read(self, node):

              t = self.revision(node)

              if not t.startswith('\1\n'):

                  return t

              s = t.index('\1\n', 2)

              return t[s + 2:]

          def _readmeta(self, node):

              t = self.revision(node)

              if not t.startswith('\1\n'):

                  return {}

              s = t.index('\1\n', 2)

              mt = t[2:s]

              m = {}

              for l in mt.splitlines():

                  k, v = l.split(": ", 1)

                  m[k] = v

              return m

          def add(self, text, meta, transaction, link, p1=None, p2=None):

              if meta or text.startswith('\1\n'):

                  mt = ["%s: %s\n" % (k, v) for k, v in sorted(meta.iteritems())]

                  text = "\1\n%s\1\n%s" % ("".join(mt), text)

              return self.addrevision(text, transaction, link, p1, p2)

          def renamed(self, node):

              if self.parents(node)[0] != revlog.nullid:

                  return False

              m = self._readmeta(node)

              if m and "copy" in m:

                  return (m["copy"], revlog.bin(m["copyrev"]))

              return False

          def size(self, rev):

              """return the size of a given revision"""

              # for revisions with renames, we have to go the slow way

              node = self.node(rev)

              if self.renamed(node):

                  return len(self.read(node))

              return revlog.revlog.size(self, rev)

          def cmp(self, node, text):

              """compare text with a given file revision"""

              # for renames, we have to go the slow way

              if text.startswith('\1\n') or self.renamed(node):

                  t2 = self.read(node)

                  return t2 != text

              return revlog.revlog.cmp(self, node, text)

	Site-wide shortcuts
/	Use quick search box
g h	Goto home page
g g	Goto my private gists page
g G	Goto my public gists page
g 0-9	Goto bookmarked items from 0-9
n r	New repository page
n g	New gist page

	Repositories
g s	Goto summary page
g c	Goto changelog page
g f	Goto files page
g F	Goto files page with file search activated
g p	Goto pull requests page
g o	Goto repository settings
g O	Goto repository access permissions settings
t s	Toggle sidebar on some pages

				# filelog.py - file history class for mercurial
				#
				# Copyright 2005-2007 Matt Mackall <mpm@selenic.com>
				#
				# This software may be used and distributed according to the terms of the
				# GNU General Public License version 2 or any later version.

				import revlog

				class filelog(revlog.revlog):
				def __init__(self, opener, path):
				revlog.revlog.__init__(self, opener,
				"/".join(("data", path + ".i")))

				def read(self, node):
				t = self.revision(node)
				if not t.startswith('\1\n'):
				return t
				s = t.index('\1\n', 2)
				return t[s + 2:]

				def _readmeta(self, node):
				t = self.revision(node)
				if not t.startswith('\1\n'):
				return {}
				s = t.index('\1\n', 2)
				mt = t[2:s]
				m = {}
				for l in mt.splitlines():
				k, v = l.split(": ", 1)
				m[k] = v
				return m

				def add(self, text, meta, transaction, link, p1=None, p2=None):
				if meta or text.startswith('\1\n'):
				mt = ["%s: %s\n" % (k, v) for k, v in sorted(meta.iteritems())]
				text = "\1\n%s\1\n%s" % ("".join(mt), text)
				return self.addrevision(text, transaction, link, p1, p2)

				def renamed(self, node):
				if self.parents(node)[0] != revlog.nullid:
				return False
				m = self._readmeta(node)
				if m and "copy" in m:
				return (m["copy"], revlog.bin(m["copyrev"]))
				return False

				def size(self, rev):
				"""return the size of a given revision"""

				# for revisions with renames, we have to go the slow way
				node = self.node(rev)
				if self.renamed(node):
				return len(self.read(node))

				return revlog.revlog.size(self, rev)

				def cmp(self, node, text):
				"""compare text with a given file revision"""

				# for renames, we have to go the slow way
				if text.startswith('\1\n') or self.renamed(node):
				t2 = self.read(node)
				return t2 != text

				return revlog.revlog.cmp(self, node, text)