upstream/mercurial-mirror Files · mercurial/pure/bdiff.py

vfs: add walk...

vfs: add walk To eliminate "path prefix" (= "the root of vfs") part from "dirpath" yielded by "os.walk()" correctly, "path prefix" should have "os.sep" at the end of own string, but it isn't easy to ensure it, because: - examination by "path.endswith(os.sep)" isn't portable Some problematic encodings use 0x5c (= "os.sep" on Windows) as the tail byte of some multi-byte characters. - "os.path.join(path, '')" isn't portable With Python 2.7.9, this invocation doesn't add "os.sep" at the end of UNC path (see issue4557 for detail). Python 2.7.9 changed also behavior of "os.path.normpath()" (see *) and "os.path.splitdrive()" for UNC path. vfs root normpath splitdrive os.sep required =============== ============== =================== ============ z:\ z:\ z: + \ no z:\foo z:\foo z: + \foo yes z:\foo\ z:\foo z: + \foo yes [before Python 2.7.9] \\foo\bar \\foo\bar '' + \\foo\bar yes \\foo\bar\ \\foo\bar (*) '' + \\foo\bar yes \\foo\bar\baz \\foo\bar\baz '' + \\foo\bar\baz yes \\foo\bar\baz\ \\foo\bar\baz '' + \\foo\bar\baz yes [Python 2.7.9] \\foo\bar \\foo\bar \\foo\bar + '' yes \\foo\bar\ \\foo\bar\ (*) \\foo\bar + \ no \\foo\bar\baz \\foo\bar\baz \\foo\bar + \baz yes \\foo\bar\baz\ \\foo\bar\baz \\foo\bar + \baz yes If it is ensured that "normpath()"-ed vfs root is passed to "splitdrive()", adding "os.sep" is required only when "path" part of "splitdrive()" result isn't "os.sep" itself. This is just what "pathutil.nameasprefix()" examines. This patch applies "os.path.normpath()" on "self.join(None)" explicitly, because it isn't ensured that vfs root is already normalized: vfs itself is constructed with "realpath=False" (= avoid normalizing in "vfs.__init__()") in many code paths. This normalization should be much cheaper than subsequent file I/O for directory traversal.

Patrick Mezard - - Load All Authors

File last commit:

r15530:eeac5e17 default


                r24725:ee751d47

default

Download file

             bdiff.py
        
                    87 lines
            
             | 2.3 KiB
            
                | text/x-python
            
             |
                PythonLexer
            
             / mercurial / pure / bdiff.py
          
                    History
                
                 |
                  Annotation
                 | Raw
                 |Copy content
                 |Copy permalink

      # bdiff.py - Python implementation of bdiff.c

      #

      # Copyright 2009 Matt Mackall <mpm@selenic.com> and others

      #

      # This software may be used and distributed according to the terms of the

      # GNU General Public License version 2 or any later version.

      import struct, difflib, re

      def splitnewlines(text):

          '''like str.splitlines, but only split on newlines.'''

          lines = [l + '\n' for l in text.split('\n')]

          if lines:

              if lines[-1] == '\n':

                  lines.pop()

              else:

                  lines[-1] = lines[-1][:-1]

          return lines

      def _normalizeblocks(a, b, blocks):

          prev = None

          r = []

          for curr in blocks:

              if prev is None:

                  prev = curr

                  continue

              shift = 0

              a1, b1, l1 = prev

              a1end = a1 + l1

              b1end = b1 + l1

              a2, b2, l2 = curr

              a2end = a2 + l2

              b2end = b2 + l2

              if a1end == a2:

                  while (a1end + shift < a2end and

                         a[a1end + shift] == b[b1end + shift]):

                      shift += 1

              elif b1end == b2:

                  while (b1end + shift < b2end and

                         a[a1end + shift] == b[b1end + shift]):

                      shift += 1

              r.append((a1, b1, l1 + shift))

              prev = a2 + shift, b2 + shift, l2 - shift

          r.append(prev)

          return r

      def bdiff(a, b):

          a = str(a).splitlines(True)

          b = str(b).splitlines(True)

          if not a:

              s = "".join(b)

              return s and (struct.pack(">lll", 0, 0, len(s)) + s)

          bin = []

          p = [0]

          for i in a: p.append(p[-1] + len(i))

          d = difflib.SequenceMatcher(None, a, b).get_matching_blocks()

          d = _normalizeblocks(a, b, d)

          la = 0

          lb = 0

          for am, bm, size in d:

              s = "".join(b[lb:bm])

              if am > la or s:

                  bin.append(struct.pack(">lll", p[la], p[am], len(s)) + s)

              la = am + size

              lb = bm + size

          return "".join(bin)

      def blocks(a, b):

          an = splitnewlines(a)

          bn = splitnewlines(b)

          d = difflib.SequenceMatcher(None, an, bn).get_matching_blocks()

          d = _normalizeblocks(an, bn, d)

          return [(i, i + n, j, j + n) for (i, j, n) in d]

      def fixws(text, allws):

          if allws:

              text = re.sub('[ \t\r]+', '', text)

          else:

              text = re.sub('[ \t\r]+', ' ', text)

              text = text.replace(' \n', '\n')

          return text

	Site-wide shortcuts
/	Use quick search box
g h	Goto home page
g g	Goto my private gists page
g G	Goto my public gists page
g 0-9	Goto bookmarked items from 0-9
n r	New repository page
n g	New gist page

	Repositories
g s	Goto summary page
g c	Goto changelog page
g f	Goto files page
g F	Goto files page with file search activated
g p	Goto pull requests page
g o	Goto repository settings
g O	Goto repository access permissions settings
t s	Toggle sidebar on some pages

				# bdiff.py - Python implementation of bdiff.c
				#
				# Copyright 2009 Matt Mackall <mpm@selenic.com> and others
				#
				# This software may be used and distributed according to the terms of the
				# GNU General Public License version 2 or any later version.

				import struct, difflib, re

				def splitnewlines(text):
				'''like str.splitlines, but only split on newlines.'''
				lines = [l + '\n' for l in text.split('\n')]
				if lines:
				if lines[-1] == '\n':
				lines.pop()
				else:
				lines[-1] = lines[-1][:-1]
				return lines

				def _normalizeblocks(a, b, blocks):
				prev = None
				r = []
				for curr in blocks:
				if prev is None:
				prev = curr
				continue
				shift = 0

				a1, b1, l1 = prev
				a1end = a1 + l1
				b1end = b1 + l1

				a2, b2, l2 = curr
				a2end = a2 + l2
				b2end = b2 + l2
				if a1end == a2:
				while (a1end + shift < a2end and
				a[a1end + shift] == b[b1end + shift]):
				shift += 1
				elif b1end == b2:
				while (b1end + shift < b2end and
				a[a1end + shift] == b[b1end + shift]):
				shift += 1
				r.append((a1, b1, l1 + shift))
				prev = a2 + shift, b2 + shift, l2 - shift
				r.append(prev)
				return r

				def bdiff(a, b):
				a = str(a).splitlines(True)
				b = str(b).splitlines(True)

				if not a:
				s = "".join(b)
				return s and (struct.pack(">lll", 0, 0, len(s)) + s)

				bin = []
				p = [0]
				for i in a: p.append(p[-1] + len(i))

				d = difflib.SequenceMatcher(None, a, b).get_matching_blocks()
				d = _normalizeblocks(a, b, d)
				la = 0
				lb = 0
				for am, bm, size in d:
				s = "".join(b[lb:bm])
				if am > la or s:
				bin.append(struct.pack(">lll", p[la], p[am], len(s)) + s)
				la = am + size
				lb = bm + size

				return "".join(bin)

				def blocks(a, b):
				an = splitnewlines(a)
				bn = splitnewlines(b)
				d = difflib.SequenceMatcher(None, an, bn).get_matching_blocks()
				d = _normalizeblocks(an, bn, d)
				return [(i, i + n, j, j + n) for (i, j, n) in d]

				def fixws(text, allws):
				if allws:
				text = re.sub('[ \t\r]+', '', text)
				else:
				text = re.sub('[ \t\r]+', ' ', text)
				text = text.replace(' \n', '\n')
				return text