/usr/share/doc/python2-docs/html/_sources/library
NameSizeModeActions
2to3.rst.txt146470644editdlrm
abc.rst.txt72190644editdlrm
aepack.rst.txt42570644editdlrm
aetools.rst.txt35320644editdlrm
aetypes.rst.txt42570644editdlrm
aifc.rst.txt70860644editdlrm
al.rst.txt53090644editdlrm
allos.rst.txt6950644editdlrm
anydbm.rst.txt41110644editdlrm
archiving.rst.txt4240644editdlrm
argparse.rst.txt746850644editdlrm
array.rst.txt105240644editdlrm
ast.rst.txt101110644editdlrm
asynchat.rst.txt92060644editdlrm
asyncore.rst.txt129370644editdlrm
atexit.rst.txt39100644editdlrm
audioop.rst.txt103950644editdlrm
autogil.rst.txt10150644editdlrm
base64.rst.txt62520644editdlrm
basehttpserver.rst.txt104040644editdlrm
bastion.rst.txt26110644editdlrm
bdb.rst.txt124560644editdlrm
binascii.rst.txt65080644editdlrm
binhex.rst.txt19100644editdlrm
bisect.rst.txt54150644editdlrm
bsddb.rst.txt75750644editdlrm
bz2.rst.txt80640644editdlrm
calendar.rst.txt112820644editdlrm
carbon.rst.txt159560644editdlrm
cd.rst.txt119740644editdlrm
cgi.rst.txt228530644editdlrm
cgihttpserver.rst.txt27880644editdlrm
cgitb.rst.txt28740644editdlrm
chunk.rst.txt49550644editdlrm
cmath.rst.txt76470644editdlrm
cmd.rst.txt85650644editdlrm
code.rst.txt71150644editdlrm
codecs.rst.txt669660644editdlrm
codeop.rst.txt37740644editdlrm
collections.rst.txt414480644editdlrm
colorpicker.rst.txt9130644editdlrm
colorsys.rst.txt18190644editdlrm
commands.rst.txt25950644editdlrm
compileall.rst.txt46780644editdlrm
compiler.rst.txt374640644editdlrm
configparser.rst.txt196070644editdlrm
constants.rst.txt23280644editdlrm
contextlib.rst.txt60120644editdlrm
cookie.rst.txt95430644editdlrm
cookielib.rst.txt278750644editdlrm
copy.rst.txt33510644editdlrm
copy_reg.rst.txt23290644editdlrm
crypt.rst.txt22920644editdlrm
crypto.rst.txt3550644editdlrm
csv.rst.txt227570644editdlrm
ctypes.rst.txt905130644editdlrm
curses.ascii.rst.txt90460644editdlrm
curses.panel.rst.txt27400644editdlrm
curses.rst.txt748770644editdlrm
custominterp.rst.txt5700644editdlrm
datatypes.rst.txt8640644editdlrm
datetime.rst.txt746790644editdlrm
dbhash.rst.txt38650644editdlrm
dbm.rst.txt31170644editdlrm
debug.rst.txt4460644editdlrm
decimal.rst.txt709260644editdlrm
development.rst.txt6400644editdlrm
difflib.rst.txt307160644editdlrm
dircache.rst.txt18130644editdlrm
dis.rst.txt232480644editdlrm
distribution.rst.txt4260644editdlrm
distutils.rst.txt19580644editdlrm
dl.rst.txt33920644editdlrm
doctest.rst.txt739850644editdlrm
docxmlrpcserver.rst.txt38020644editdlrm
dumbdbm.rst.txt28410644editdlrm
dummy_thread.rst.txt10580644editdlrm
dummy_threading.rst.txt7990644editdlrm
easydialogs.rst.txt103460644editdlrm
email-examples.rst.txt12710644editdlrm
email.charset.rst.txt96540644editdlrm
email.encoders.rst.txt23760644editdlrm
email.errors.rst.txt40060644editdlrm
email.generator.rst.txt61320644editdlrm
email.header.rst.txt75310644editdlrm
email.iterators.rst.txt24210644editdlrm
email.message.rst.txt252170644editdlrm
email.mime.rst.txt99170644editdlrm
email.parser.rst.txt103270644editdlrm
email.rst.txt161110644editdlrm
email.utils.rst.txt64750644editdlrm
ensurepip.rst.txt50220644editdlrm
errno.rst.txt67080644editdlrm
exceptions.rst.txt183430644editdlrm
fcntl.rst.txt73610644editdlrm
filecmp.rst.txt53480644editdlrm
fileformats.rst.txt3020644editdlrm
fileinput.rst.txt74170644editdlrm
filesys.rst.txt8060644editdlrm
fl.rst.txt176460644editdlrm
fm.rst.txt26990644editdlrm
fnmatch.rst.txt31040644editdlrm
formatter.rst.txt132450644editdlrm
fpectl.rst.txt41710644editdlrm
fpformat.rst.txt17470644editdlrm
fractions.rst.txt52970644editdlrm
framework.rst.txt114440644editdlrm
frameworks.rst.txt3780644editdlrm
ftplib.rst.txt157200644editdlrm
functions.rst.txt755540644editdlrm
functools.rst.txt74480644editdlrm
future_builtins.rst.txt20100644editdlrm
gc.rst.txt90140644editdlrm
gdbm.rst.txt48820644editdlrm
gensuitemodule.rst.txt31130644editdlrm
getopt.rst.txt66700644editdlrm
getpass.rst.txt18760644editdlrm
gettext.rst.txt290400644editdlrm
gl.rst.txt60090644editdlrm
glob.rst.txt24140644editdlrm
grp.rst.txt22560644editdlrm
gzip.rst.txt48280644editdlrm
hashlib.rst.txt73790644editdlrm
heapq.rst.txt131850644editdlrm
hmac.rst.txt30730644editdlrm
hotshot.rst.txt42890644editdlrm
htmllib.rst.txt73790644editdlrm
htmlparser.rst.txt116400644editdlrm
httplib.rst.txt374540644editdlrm
i18n.rst.txt4090644editdlrm
ic.rst.txt50050644editdlrm
idle.rst.txt221820644editdlrm
imageop.rst.txt40010644editdlrm
imaplib.rst.txt172210644editdlrm
imgfile.rst.txt27650644editdlrm
imghdr.rst.txt26350644editdlrm
imp.rst.txt125930644editdlrm
importlib.rst.txt11260644editdlrm
imputil.rst.txt70230644editdlrm
index.rst.txt22870644editdlrm
inspect.rst.txt281510644editdlrm
internet.rst.txt9500644editdlrm
intro.rst.txt28030644editdlrm
io.rst.txt390150644editdlrm
ipc.rst.txt6300644editdlrm
itertools.rst.txt363240644editdlrm
jpeg.rst.txt38580644editdlrm
json.rst.txt255450644editdlrm
keyword.rst.txt6170644editdlrm
language.rst.txt5230644editdlrm
linecache.rst.txt18870644editdlrm
locale.rst.txt249750644editdlrm
logging.config.rst.txt315600644editdlrm
logging.handlers.rst.txt281530644editdlrm
logging.rst.txt468820644editdlrm
mac.rst.txt7910644editdlrm
macos.rst.txt38240644editdlrm
macosa.rst.txt39640644editdlrm
macostools.rst.txt40170644editdlrm
macpath.rst.txt6500644editdlrm
mailbox.rst.txt681290644editdlrm
mailcap.rst.txt36750644editdlrm
markup.rst.txt12490644editdlrm
marshal.rst.txt56890644editdlrm
math.rst.txt109200644editdlrm
md5.rst.txt28150644editdlrm
mhlib.rst.txt39660644editdlrm
mimetools.rst.txt45040644editdlrm
mimetypes.rst.txt98390644editdlrm
mimewriter.rst.txt33600644editdlrm
mimify.rst.txt35190644editdlrm
miniaeframe.rst.txt25640644editdlrm
misc.rst.txt2480644editdlrm
mm.rst.txt4470644editdlrm
mmap.rst.txt104810644editdlrm
modulefinder.rst.txt33800644editdlrm
modules.rst.txt3820644editdlrm
msilib.rst.txt190620644editdlrm
msvcrt.rst.txt43460644editdlrm
multifile.rst.txt66130644editdlrm
multiprocessing.rst.txt926210644editdlrm
mutex.rst.txt19390644editdlrm
netdata.rst.txt4320644editdlrm
netrc.rst.txt31190644editdlrm
new.rst.txt26530644editdlrm
nis.rst.txt21080644editdlrm
nntplib.rst.txt145310644editdlrm
numbers.rst.txt80070644editdlrm
numeric.rst.txt7510644editdlrm
operator.rst.txt220920644editdlrm
optparse.rst.txt771010644editdlrm
os.path.rst.txt130940644editdlrm
os.rst.txt825800644editdlrm
ossaudiodev.rst.txt173090644editdlrm
othergui.rst.txt27210644editdlrm
parser.rst.txt153850644editdlrm
pdb.rst.txt160430644editdlrm
persistence.rst.txt8260644editdlrm
pickle.rst.txt372620644editdlrm
pickletools.rst.txt19970644editdlrm
pipes.rst.txt37860644editdlrm
pkgutil.rst.txt77140644editdlrm
platform.rst.txt95540644editdlrm
plistlib.rst.txt41340644editdlrm
popen2.rst.txt70220644editdlrm
poplib.rst.txt62200644editdlrm
posix.rst.txt36300644editdlrm
posixfile.rst.txt72000644editdlrm
pprint.rst.txt90710644editdlrm
profile.rst.txt286660644editdlrm
pty.rst.txt17620644editdlrm
pwd.rst.txt27250644editdlrm
pyclbr.rst.txt32960644editdlrm
pydoc.rst.txt40830644editdlrm
pyexpat.rst.txt288910644editdlrm
python.rst.txt5140644editdlrm
py_compile.rst.txt24790644editdlrm
queue.rst.txt70340644editdlrm
quopri.rst.txt26700644editdlrm
random.rst.txt133190644editdlrm
re.rst.txt551020644editdlrm
readline.rst.txt105130644editdlrm
repr.rst.txt47180644editdlrm
resource.rst.txt95950644editdlrm
restricted.rst.txt33270644editdlrm
rexec.rst.txt117420644editdlrm
rfc822.rst.txt140370644editdlrm
rlcompleter.rst.txt24940644editdlrm
robotparser.rst.txt21900644editdlrm
runpy.rst.txt69380644editdlrm
sched.rst.txt46440644editdlrm
scrolledtext.rst.txt13790644editdlrm
select.rst.txt207050644editdlrm
sets.rst.txt149730644editdlrm
sgi.rst.txt3220644editdlrm
sgmllib.rst.txt106640644editdlrm
sha.rst.txt28070644editdlrm
shelve.rst.txt83190644editdlrm
shlex.rst.txt113130644editdlrm
shutil.rst.txt134910644editdlrm
signal.rst.txt107190644editdlrm
simplehttpserver.rst.txt45590644editdlrm
simplexmlrpcserver.rst.txt108810644editdlrm
site.rst.txt78800644editdlrm
smtpd.rst.txt24640644editdlrm
smtplib.rst.txt149060644editdlrm
sndhdr.rst.txt17590644editdlrm
socket.rst.txt407650644editdlrm
socketserver.rst.txt225120644editdlrm
someos.rst.txt5990644editdlrm
spwd.rst.txt28250644editdlrm
sqlite3.rst.txt354480644editdlrm
ssl.rst.txt765540644editdlrm
stat.rst.txt77780644editdlrm
statvfs.rst.txt13000644editdlrm
stdtypes.rst.txt1226680644editdlrm
string.rst.txt440850644editdlrm
stringio.rst.txt41660644editdlrm
stringprep.rst.txt42380644editdlrm
strings.rst.txt7460644editdlrm
struct.rst.txt171000644editdlrm
subprocess.rst.txt336480644editdlrm
sun.rst.txt2490644editdlrm
sunau.rst.txt71240644editdlrm
sunaudio.rst.txt58490644editdlrm
symbol.rst.txt9750644editdlrm
symtable.rst.txt50640644editdlrm
sys.rst.txt475200644editdlrm
sysconfig.rst.txt76390644editdlrm
syslog.rst.txt39310644editdlrm
tabnanny.rst.txt19990644editdlrm
tarfile.rst.txt282140644editdlrm
telnetlib.rst.txt74820644editdlrm
tempfile.rst.txt105040644editdlrm
termios.rst.txt37330644editdlrm
test.rst.txt176590644editdlrm
textwrap.rst.txt85520644editdlrm
thread.rst.txt65880644editdlrm
threading.rst.txt324680644editdlrm
time.rst.txt260200644editdlrm
timeit.rst.txt115980644editdlrm
tix.rst.txt226930644editdlrm
tk.rst.txt16120644editdlrm
tkinter.rst.txt334560644editdlrm
token.rst.txt24510644editdlrm
tokenize.rst.txt55700644editdlrm
trace.rst.txt67240644editdlrm
traceback.rst.txt107110644editdlrm
ttk.rst.txt574110644editdlrm
tty.rst.txt10110644editdlrm
turtle.rst.txt640970644editdlrm
types.rst.txt62000644editdlrm
undoc.rst.txt65500644editdlrm
unicodedata.rst.txt57360644editdlrm
unittest.rst.txt829520644editdlrm
unix.rst.txt4900644editdlrm
urllib.rst.txt249600644editdlrm
urllib2.rst.txt352110644editdlrm
urlparse.rst.txt160710644editdlrm
user.rst.txt27480644editdlrm
userdict.rst.txt92900644editdlrm
uu.rst.txt23690644editdlrm
uuid.rst.txt83780644editdlrm
warnings.rst.txt198150644editdlrm
wave.rst.txt50470644editdlrm
weakref.rst.txt129260644editdlrm
webbrowser.rst.txt98630644editdlrm
whichdb.rst.txt9310644editdlrm
windows.rst.txt2730644editdlrm
winsound.rst.txt50660644editdlrm
wsgiref.rst.txt305670644editdlrm
xdrlib.rst.txt80770644editdlrm
xml.dom.minidom.rst.txt112080644editdlrm
xml.dom.pulldom.rst.txt15710644editdlrm
xml.dom.rst.txt401820644editdlrm
xml.etree.elementtree.rst.txt357070644editdlrm
xml.rst.txt60900644editdlrm
xml.sax.handler.rst.txt153710644editdlrm
xml.sax.reader.rst.txt122570644editdlrm
xml.sax.rst.txt64600644editdlrm
xml.sax.utils.rst.txt35600644editdlrm
xmlrpclib.rst.txt225590644editdlrm
zipfile.rst.txt188710644editdlrm
zipimport.rst.txt59570644editdlrm
zlib.rst.txt127660644editdlrm
_winreg.rst.txt233050644editdlrm
__builtin__.rst.txt14940644editdlrm
__future__.rst.txt49520644editdlrm
__main__.rst.txt5350644editdlrm
Edit: /usr/share/doc/python2-docs/html/_sources/library/difflib.rst.txt (30716B)
:mod:`difflib` --- Helpers for computing deltas =============================================== .. module:: difflib :synopsis: Helpers for computing differences between objects. .. moduleauthor:: Tim Peters .. sectionauthor:: Tim Peters .. Markup by Fred L. Drake, Jr. .. testsetup:: import sys from difflib import * .. versionadded:: 2.1 This module provides classes and functions for comparing sequences. It can be used for example, for comparing files, and can produce difference information in various formats, including HTML and context and unified diffs. For comparing directories and files, see also, the :mod:`filecmp` module. .. class:: SequenceMatcher This is a flexible class for comparing pairs of sequences of any type, so long as the sequence elements are :term:`hashable`. The basic algorithm predates, and is a little fancier than, an algorithm published in the late 1980's by Ratcliff and Obershelp under the hyperbolic name "gestalt pattern matching." The idea is to find the longest contiguous matching subsequence that contains no "junk" elements (the Ratcliff and Obershelp algorithm doesn't address junk). The same idea is then applied recursively to the pieces of the sequences to the left and to the right of the matching subsequence. This does not yield minimal edit sequences, but does tend to yield matches that "look right" to people. **Timing:** The basic Ratcliff-Obershelp algorithm is cubic time in the worst case and quadratic time in the expected case. :class:`SequenceMatcher` is quadratic time for the worst case and has expected-case behavior dependent in a complicated way on how many elements the sequences have in common; best case time is linear. **Automatic junk heuristic:** :class:`SequenceMatcher` supports a heuristic that automatically treats certain sequence items as junk. The heuristic counts how many times each individual item appears in the sequence. If an item's duplicates (after the first one) account for more than 1% of the sequence and the sequence is at least 200 items long, this item is marked as "popular" and is treated as junk for the purpose of sequence matching. This heuristic can be turned off by setting the ``autojunk`` argument to ``False`` when creating the :class:`SequenceMatcher`. .. versionadded:: 2.7.1 The *autojunk* parameter. .. class:: Differ This is a class for comparing sequences of lines of text, and producing human-readable differences or deltas. Differ uses :class:`SequenceMatcher` both to compare sequences of lines, and to compare sequences of characters within similar (near-matching) lines. Each line of a :class:`Differ` delta begins with a two-letter code: +----------+-------------------------------------------+ | Code | Meaning | +==========+===========================================+ | ``'- '`` | line unique to sequence 1 | +----------+-------------------------------------------+ | ``'+ '`` | line unique to sequence 2 | +----------+-------------------------------------------+ | ``' '`` | line common to both sequences | +----------+-------------------------------------------+ | ``'? '`` | line not present in either input sequence | +----------+-------------------------------------------+ Lines beginning with '``?``' attempt to guide the eye to intraline differences, and were not present in either input sequence. These lines can be confusing if the sequences contain tab characters. .. class:: HtmlDiff This class can be used to create an HTML table (or a complete HTML file containing the table) showing a side by side, line by line comparison of text with inter-line and intra-line change highlights. The table can be generated in either full or contextual difference mode. The constructor for this class is: .. function:: __init__(tabsize=8, wrapcolumn=None, linejunk=None, charjunk=IS_CHARACTER_JUNK) Initializes instance of :class:`HtmlDiff`. *tabsize* is an optional keyword argument to specify tab stop spacing and defaults to ``8``. *wrapcolumn* is an optional keyword to specify column number where lines are broken and wrapped, defaults to ``None`` where lines are not wrapped. *linejunk* and *charjunk* are optional keyword arguments passed into :func:`ndiff` (used by :class:`HtmlDiff` to generate the side by side HTML differences). See :func:`ndiff` documentation for argument default values and descriptions. The following methods are public: .. function:: make_file(fromlines, tolines [, fromdesc][, todesc][, context][, numlines]) Compares *fromlines* and *tolines* (lists of strings) and returns a string which is a complete HTML file containing a table showing line by line differences with inter-line and intra-line changes highlighted. *fromdesc* and *todesc* are optional keyword arguments to specify from/to file column header strings (both default to an empty string). *context* and *numlines* are both optional keyword arguments. Set *context* to ``True`` when contextual differences are to be shown, else the default is ``False`` to show the full files. *numlines* defaults to ``5``. When *context* is ``True`` *numlines* controls the number of context lines which surround the difference highlights. When *context* is ``False`` *numlines* controls the number of lines which are shown before a difference highlight when using the "next" hyperlinks (setting to zero would cause the "next" hyperlinks to place the next difference highlight at the top of the browser without any leading context). .. function:: make_table(fromlines, tolines [, fromdesc][, todesc][, context][, numlines]) Compares *fromlines* and *tolines* (lists of strings) and returns a string which is a complete HTML table showing line by line differences with inter-line and intra-line changes highlighted. The arguments for this method are the same as those for the :meth:`make_file` method. :file:`Tools/scripts/diff.py` is a command-line front-end to this class and contains a good example of its use. .. versionadded:: 2.4 .. function:: context_diff(a, b[, fromfile][, tofile][, fromfiledate][, tofiledate][, n][, lineterm]) Compare *a* and *b* (lists of strings); return a delta (a :term:`generator` generating the delta lines) in context diff format. Context diffs are a compact way of showing just the lines that have changed plus a few lines of context. The changes are shown in a before/after style. The number of context lines is set by *n* which defaults to three. By default, the diff control lines (those with ``***`` or ``---``) are created with a trailing newline. This is helpful so that inputs created from :func:`file.readlines` result in diffs that are suitable for use with :func:`file.writelines` since both the inputs and outputs have trailing newlines. For inputs that do not have trailing newlines, set the *lineterm* argument to ``""`` so that the output will be uniformly newline free. The context diff format normally has a header for filenames and modification times. Any or all of these may be specified using strings for *fromfile*, *tofile*, *fromfiledate*, and *tofiledate*. The modification times are normally expressed in the ISO 8601 format. If not specified, the strings default to blanks. >>> s1 = ['bacon\n', 'eggs\n', 'ham\n', 'guido\n'] >>> s2 = ['python\n', 'eggy\n', 'hamster\n', 'guido\n'] >>> for line in context_diff(s1, s2, fromfile='before.py', tofile='after.py'): ... sys.stdout.write(line) # doctest: +NORMALIZE_WHITESPACE *** before.py --- after.py *************** *** 1,4 **** ! bacon ! eggs ! ham guido --- 1,4 ---- ! python ! eggy ! hamster guido See :ref:`difflib-interface` for a more detailed example. .. versionadded:: 2.3 .. function:: get_close_matches(word, possibilities[, n][, cutoff]) Return a list of the best "good enough" matches. *word* is a sequence for which close matches are desired (typically a string), and *possibilities* is a list of sequences against which to match *word* (typically a list of strings). Optional argument *n* (default ``3``) is the maximum number of close matches to return; *n* must be greater than ``0``. Optional argument *cutoff* (default ``0.6``) is a float in the range [0, 1]. Possibilities that don't score at least that similar to *word* are ignored. The best (no more than *n*) matches among the possibilities are returned in a list, sorted by similarity score, most similar first. >>> get_close_matches('appel', ['ape', 'apple', 'peach', 'puppy']) ['apple', 'ape'] >>> import keyword >>> get_close_matches('wheel', keyword.kwlist) ['while'] >>> get_close_matches('apple', keyword.kwlist) [] >>> get_close_matches('accept', keyword.kwlist) ['except'] .. function:: ndiff(a, b[, linejunk][, charjunk]) Compare *a* and *b* (lists of strings); return a :class:`Differ`\ -style delta (a :term:`generator` generating the delta lines). Optional keyword parameters *linejunk* and *charjunk* are for filter functions (or ``None``): *linejunk*: A function that accepts a single string argument, and returns true if the string is junk, or false if not. The default is (``None``), starting with Python 2.3. Before then, the default was the module-level function :func:`IS_LINE_JUNK`, which filters out lines without visible characters, except for at most one pound character (``'#'``). As of Python 2.3, the underlying :class:`SequenceMatcher` class does a dynamic analysis of which lines are so frequent as to constitute noise, and this usually works better than the pre-2.3 default. *charjunk*: A function that accepts a character (a string of length 1), and returns if the character is junk, or false if not. The default is module-level function :func:`IS_CHARACTER_JUNK`, which filters out whitespace characters (a blank or tab; note: bad idea to include newline in this!). :file:`Tools/scripts/ndiff.py` is a command-line front-end to this function. >>> diff = ndiff('one\ntwo\nthree\n'.splitlines(1), ... 'ore\ntree\nemu\n'.splitlines(1)) >>> print ''.join(diff), - one ? ^ + ore ? ^ - two - three ? - + tree + emu .. function:: restore(sequence, which) Return one of the two sequences that generated a delta. Given a *sequence* produced by :meth:`Differ.compare` or :func:`ndiff`, extract lines originating from file 1 or 2 (parameter *which*), stripping off line prefixes. Example: >>> diff = ndiff('one\ntwo\nthree\n'.splitlines(1), ... 'ore\ntree\nemu\n'.splitlines(1)) >>> diff = list(diff) # materialize the generated delta into a list >>> print ''.join(restore(diff, 1)), one two three >>> print ''.join(restore(diff, 2)), ore tree emu .. function:: unified_diff(a, b[, fromfile][, tofile][, fromfiledate][, tofiledate][, n][, lineterm]) Compare *a* and *b* (lists of strings); return a delta (a :term:`generator` generating the delta lines) in unified diff format. Unified diffs are a compact way of showing just the lines that have changed plus a few lines of context. The changes are shown in an inline style (instead of separate before/after blocks). The number of context lines is set by *n* which defaults to three. By default, the diff control lines (those with ``---``, ``+++``, or ``@@``) are created with a trailing newline. This is helpful so that inputs created from :func:`file.readlines` result in diffs that are suitable for use with :func:`file.writelines` since both the inputs and outputs have trailing newlines. For inputs that do not have trailing newlines, set the *lineterm* argument to ``""`` so that the output will be uniformly newline free. The context diff format normally has a header for filenames and modification times. Any or all of these may be specified using strings for *fromfile*, *tofile*, *fromfiledate*, and *tofiledate*. The modification times are normally expressed in the ISO 8601 format. If not specified, the strings default to blanks. >>> s1 = ['bacon\n', 'eggs\n', 'ham\n', 'guido\n'] >>> s2 = ['python\n', 'eggy\n', 'hamster\n', 'guido\n'] >>> for line in unified_diff(s1, s2, fromfile='before.py', tofile='after.py'): ... sys.stdout.write(line) # doctest: +NORMALIZE_WHITESPACE --- before.py +++ after.py @@ -1,4 +1,4 @@ -bacon -eggs -ham +python +eggy +hamster guido See :ref:`difflib-interface` for a more detailed example. .. versionadded:: 2.3 .. function:: IS_LINE_JUNK(line) Return true for ignorable lines. The line *line* is ignorable if *line* is blank or contains a single ``'#'``, otherwise it is not ignorable. Used as a default for parameter *linejunk* in :func:`ndiff` before Python 2.3. .. function:: IS_CHARACTER_JUNK(ch) Return true for ignorable characters. The character *ch* is ignorable if *ch* is a space or tab, otherwise it is not ignorable. Used as a default for parameter *charjunk* in :func:`ndiff`. .. seealso:: `Pattern Matching: The Gestalt Approach `_ Discussion of a similar algorithm by John W. Ratcliff and D. E. Metzener. This was published in `Dr. Dobb's Journal `_ in July, 1988. .. _sequence-matcher: SequenceMatcher Objects ----------------------- The :class:`SequenceMatcher` class has this constructor: .. class:: SequenceMatcher(isjunk=None, a='', b='', autojunk=True) Optional argument *isjunk* must be ``None`` (the default) or a one-argument function that takes a sequence element and returns true if and only if the element is "junk" and should be ignored. Passing ``None`` for *isjunk* is equivalent to passing ``lambda x: 0``; in other words, no elements are ignored. For example, pass:: lambda x: x in " \t" if you're comparing lines as sequences of characters, and don't want to synch up on blanks or hard tabs. The optional arguments *a* and *b* are sequences to be compared; both default to empty strings. The elements of both sequences must be :term:`hashable`. The optional argument *autojunk* can be used to disable the automatic junk heuristic. .. versionadded:: 2.7.1 The *autojunk* parameter. :class:`SequenceMatcher` objects have the following methods: .. method:: set_seqs(a, b) Set the two sequences to be compared. :class:`SequenceMatcher` computes and caches detailed information about the second sequence, so if you want to compare one sequence against many sequences, use :meth:`set_seq2` to set the commonly used sequence once and call :meth:`set_seq1` repeatedly, once for each of the other sequences. .. method:: set_seq1(a) Set the first sequence to be compared. The second sequence to be compared is not changed. .. method:: set_seq2(b) Set the second sequence to be compared. The first sequence to be compared is not changed. .. method:: find_longest_match(alo, ahi, blo, bhi) Find longest matching block in ``a[alo:ahi]`` and ``b[blo:bhi]``. If *isjunk* was omitted or ``None``, :meth:`find_longest_match` returns ``(i, j, k)`` such that ``a[i:i+k]`` is equal to ``b[j:j+k]``, where ``alo <= i <= i+k <= ahi`` and ``blo <= j <= j+k <= bhi``. For all ``(i', j', k')`` meeting those conditions, the additional conditions ``k >= k'``, ``i <= i'``, and if ``i == i'``, ``j <= j'`` are also met. In other words, of all maximal matching blocks, return one that starts earliest in *a*, and of all those maximal matching blocks that start earliest in *a*, return the one that starts earliest in *b*. >>> s = SequenceMatcher(None, " abcd", "abcd abcd") >>> s.find_longest_match(0, 5, 0, 9) Match(a=0, b=4, size=5) If *isjunk* was provided, first the longest matching block is determined as above, but with the additional restriction that no junk element appears in the block. Then that block is extended as far as possible by matching (only) junk elements on both sides. So the resulting block never matches on junk except as identical junk happens to be adjacent to an interesting match. Here's the same example as before, but considering blanks to be junk. That prevents ``' abcd'`` from matching the ``' abcd'`` at the tail end of the second sequence directly. Instead only the ``'abcd'`` can match, and matches the leftmost ``'abcd'`` in the second sequence: >>> s = SequenceMatcher(lambda x: x==" ", " abcd", "abcd abcd") >>> s.find_longest_match(0, 5, 0, 9) Match(a=1, b=0, size=4) If no blocks match, this returns ``(alo, blo, 0)``. .. versionchanged:: 2.6 This method returns a :term:`named tuple` ``Match(a, b, size)``. .. method:: get_matching_blocks() Return list of triples describing non-overlapping matching subsequences. Each triple is of the form ``(i, j, n)``, and means that ``a[i:i+n] == b[j:j+n]``. The triples are monotonically increasing in *i* and *j*. The last triple is a dummy, and has the value ``(len(a), len(b), 0)``. It is the only triple with ``n == 0``. If ``(i, j, n)`` and ``(i', j', n')`` are adjacent triples in the list, and the second is not the last triple in the list, then ``i+n < i'`` or ``j+n < j'``; in other words, adjacent triples always describe non-adjacent equal blocks. .. XXX Explain why a dummy is used! .. versionchanged:: 2.5 The guarantee that adjacent triples always describe non-adjacent blocks was implemented. .. doctest:: >>> s = SequenceMatcher(None, "abxcd", "abcd") >>> s.get_matching_blocks() [Match(a=0, b=0, size=2), Match(a=3, b=2, size=2), Match(a=5, b=4, size=0)] .. method:: get_opcodes() Return list of 5-tuples describing how to turn *a* into *b*. Each tuple is of the form ``(tag, i1, i2, j1, j2)``. The first tuple has ``i1 == j1 == 0``, and remaining tuples have *i1* equal to the *i2* from the preceding tuple, and, likewise, *j1* equal to the previous *j2*. The *tag* values are strings, with these meanings: +---------------+---------------------------------------------+ | Value | Meaning | +===============+=============================================+ | ``'replace'`` | ``a[i1:i2]`` should be replaced by | | | ``b[j1:j2]``. | +---------------+---------------------------------------------+ | ``'delete'`` | ``a[i1:i2]`` should be deleted. Note that | | | ``j1 == j2`` in this case. | +---------------+---------------------------------------------+ | ``'insert'`` | ``b[j1:j2]`` should be inserted at | | | ``a[i1:i1]``. Note that ``i1 == i2`` in | | | this case. | +---------------+---------------------------------------------+ | ``'equal'`` | ``a[i1:i2] == b[j1:j2]`` (the sub-sequences | | | are equal). | +---------------+---------------------------------------------+ For example: >>> a = "qabxcd" >>> b = "abycdf" >>> s = SequenceMatcher(None, a, b) >>> for tag, i1, i2, j1, j2 in s.get_opcodes(): ... print ("%7s a[%d:%d] (%s) b[%d:%d] (%s)" % ... (tag, i1, i2, a[i1:i2], j1, j2, b[j1:j2])) delete a[0:1] (q) b[0:0] () equal a[1:3] (ab) b[0:2] (ab) replace a[3:4] (x) b[2:3] (y) equal a[4:6] (cd) b[3:5] (cd) insert a[6:6] () b[5:6] (f) .. method:: get_grouped_opcodes([n]) Return a :term:`generator` of groups with up to *n* lines of context. Starting with the groups returned by :meth:`get_opcodes`, this method splits out smaller change clusters and eliminates intervening ranges which have no changes. The groups are returned in the same format as :meth:`get_opcodes`. .. versionadded:: 2.3 .. method:: ratio() Return a measure of the sequences' similarity as a float in the range [0, 1]. Where T is the total number of elements in both sequences, and M is the number of matches, this is 2.0\*M / T. Note that this is ``1.0`` if the sequences are identical, and ``0.0`` if they have nothing in common. This is expensive to compute if :meth:`get_matching_blocks` or :meth:`get_opcodes` hasn't already been called, in which case you may want to try :meth:`quick_ratio` or :meth:`real_quick_ratio` first to get an upper bound. .. method:: quick_ratio() Return an upper bound on :meth:`ratio` relatively quickly. .. method:: real_quick_ratio() Return an upper bound on :meth:`ratio` very quickly. The three methods that return the ratio of matching to total characters can give different results due to differing levels of approximation, although :meth:`quick_ratio` and :meth:`real_quick_ratio` are always at least as large as :meth:`ratio`: >>> s = SequenceMatcher(None, "abcd", "bcde") >>> s.ratio() 0.75 >>> s.quick_ratio() 0.75 >>> s.real_quick_ratio() 1.0 .. _sequencematcher-examples: SequenceMatcher Examples ------------------------ This example compares two strings, considering blanks to be "junk:" >>> s = SequenceMatcher(lambda x: x == " ", ... "private Thread currentThread;", ... "private volatile Thread currentThread;") :meth:`ratio` returns a float in [0, 1], measuring the similarity of the sequences. As a rule of thumb, a :meth:`ratio` value over 0.6 means the sequences are close matches: >>> print round(s.ratio(), 3) 0.866 If you're only interested in where the sequences match, :meth:`get_matching_blocks` is handy: >>> for block in s.get_matching_blocks(): ... print "a[%d] and b[%d] match for %d elements" % block a[0] and b[0] match for 8 elements a[8] and b[17] match for 21 elements a[29] and b[38] match for 0 elements Note that the last tuple returned by :meth:`get_matching_blocks` is always a dummy, ``(len(a), len(b), 0)``, and this is the only case in which the last tuple element (number of elements matched) is ``0``. If you want to know how to change the first sequence into the second, use :meth:`get_opcodes`: >>> for opcode in s.get_opcodes(): ... print "%6s a[%d:%d] b[%d:%d]" % opcode equal a[0:8] b[0:8] insert a[8:8] b[8:17] equal a[8:29] b[17:38] .. seealso:: * The :func:`get_close_matches` function in this module which shows how simple code building on :class:`SequenceMatcher` can be used to do useful work. * `Simple version control recipe `_ for a small application built with :class:`SequenceMatcher`. .. _differ-objects: Differ Objects -------------- Note that :class:`Differ`\ -generated deltas make no claim to be **minimal** diffs. To the contrary, minimal diffs are often counter-intuitive, because they synch up anywhere possible, sometimes accidental matches 100 pages apart. Restricting synch points to contiguous matches preserves some notion of locality, at the occasional cost of producing a longer diff. The :class:`Differ` class has this constructor: .. class:: Differ([linejunk[, charjunk]]) Optional keyword parameters *linejunk* and *charjunk* are for filter functions (or ``None``): *linejunk*: A function that accepts a single string argument, and returns true if the string is junk. The default is ``None``, meaning that no line is considered junk. *charjunk*: A function that accepts a single character argument (a string of length 1), and returns true if the character is junk. The default is ``None``, meaning that no character is considered junk. :class:`Differ` objects are used (deltas generated) via a single method: .. method:: Differ.compare(a, b) Compare two sequences of lines, and generate the delta (a sequence of lines). Each sequence must contain individual single-line strings ending with newlines. Such sequences can be obtained from the :meth:`~file.readlines` method of file-like objects. The delta generated also consists of newline-terminated strings, ready to be printed as-is via the :meth:`~file.writelines` method of a file-like object. .. _differ-examples: Differ Example -------------- This example compares two texts. First we set up the texts, sequences of individual single-line strings ending with newlines (such sequences can also be obtained from the :meth:`~file.readlines` method of file-like objects): >>> text1 = ''' 1. Beautiful is better than ugly. ... 2. Explicit is better than implicit. ... 3. Simple is better than complex. ... 4. Complex is better than complicated. ... '''.splitlines(1) >>> len(text1) 4 >>> text1[0][-1] '\n' >>> text2 = ''' 1. Beautiful is better than ugly. ... 3. Simple is better than complex. ... 4. Complicated is better than complex. ... 5. Flat is better than nested. ... '''.splitlines(1) Next we instantiate a Differ object: >>> d = Differ() Note that when instantiating a :class:`Differ` object we may pass functions to filter out line and character "junk." See the :meth:`Differ` constructor for details. Finally, we compare the two: >>> result = list(d.compare(text1, text2)) ``result`` is a list of strings, so let's pretty-print it: >>> from pprint import pprint >>> pprint(result) [' 1. Beautiful is better than ugly.\n', '- 2. Explicit is better than implicit.\n', '- 3. Simple is better than complex.\n', '+ 3. Simple is better than complex.\n', '? ++\n', '- 4. Complex is better than complicated.\n', '? ^ ---- ^\n', '+ 4. Complicated is better than complex.\n', '? ++++ ^ ^\n', '+ 5. Flat is better than nested.\n'] As a single multi-line string it looks like this: >>> import sys >>> sys.stdout.writelines(result) 1. Beautiful is better than ugly. - 2. Explicit is better than implicit. - 3. Simple is better than complex. + 3. Simple is better than complex. ? ++ - 4. Complex is better than complicated. ? ^ ---- ^ + 4. Complicated is better than complex. ? ++++ ^ ^ + 5. Flat is better than nested. .. _difflib-interface: A command-line interface to difflib ----------------------------------- This example shows how to use difflib to create a ``diff``-like utility. It is also contained in the Python source distribution, as :file:`Tools/scripts/diff.py`. .. testcode:: """ Command line interface to difflib.py providing diffs in four formats: * ndiff: lists every line and highlights interline changes. * context: highlights clusters of changes in a before/after format. * unified: highlights clusters of changes in an inline format. * html: generates side by side comparison with change highlights. """ import sys, os, time, difflib, optparse def main(): # Configure the option parser usage = "usage: %prog [options] fromfile tofile" parser = optparse.OptionParser(usage) parser.add_option("-c", action="store_true", default=False, help='Produce a context format diff (default)') parser.add_option("-u", action="store_true", default=False, help='Produce a unified format diff') hlp = 'Produce HTML side by side diff (can use -c and -l in conjunction)' parser.add_option("-m", action="store_true", default=False, help=hlp) parser.add_option("-n", action="store_true", default=False, help='Produce a ndiff format diff') parser.add_option("-l", "--lines", type="int", default=3, help='Set number of context lines (default 3)') (options, args) = parser.parse_args() if len(args) == 0: parser.print_help() sys.exit(1) if len(args) != 2: parser.error("need to specify both a fromfile and tofile") n = options.lines fromfile, tofile = args # as specified in the usage string # we're passing these as arguments to the diff function fromdate = time.ctime(os.stat(fromfile).st_mtime) todate = time.ctime(os.stat(tofile).st_mtime) with open(fromfile, 'U') as f: fromlines = f.readlines() with open(tofile, 'U') as f: tolines = f.readlines() if options.u: diff = difflib.unified_diff(fromlines, tolines, fromfile, tofile, fromdate, todate, n=n) elif options.n: diff = difflib.ndiff(fromlines, tolines) elif options.m: diff = difflib.HtmlDiff().make_file(fromlines, tolines, fromfile, tofile, context=options.c, numlines=n) else: diff = difflib.context_diff(fromlines, tolines, fromfile, tofile, fromdate, todate, n=n) # we're using writelines because diff is a generator sys.stdout.writelines(diff) if __name__ == '__main__': main()