Re: [PATCH] Add notmuch-remove-duplicates.py script to contrib.
authorDmitry Kurochkin <dmitry.kurochkin@gmail.com>
Tue, 4 Sep 2012 20:12:39 +0000 (00:12 +0400)
committerW. Trevor King <wking@tremily.us>
Fri, 7 Nov 2014 17:49:22 +0000 (09:49 -0800)
3d/e9948133aef2d1ca93625ff15818ac11380ef2 [new file with mode: 0644]

diff --git a/3d/e9948133aef2d1ca93625ff15818ac11380ef2 b/3d/e9948133aef2d1ca93625ff15818ac11380ef2
new file mode 100644 (file)
index 0000000..23f428e
--- /dev/null
@@ -0,0 +1,382 @@
+Return-Path: <dmitry.kurochkin@gmail.com>\r
+X-Original-To: notmuch@notmuchmail.org\r
+Delivered-To: notmuch@notmuchmail.org\r
+Received: from localhost (localhost [127.0.0.1])\r
+       by olra.theworths.org (Postfix) with ESMTP id 6E5B8431FB6\r
+       for <notmuch@notmuchmail.org>; Tue,  4 Sep 2012 13:12:45 -0700 (PDT)\r
+X-Virus-Scanned: Debian amavisd-new at olra.theworths.org\r
+X-Spam-Flag: NO\r
+X-Spam-Score: -0.799\r
+X-Spam-Level: \r
+X-Spam-Status: No, score=-0.799 tagged_above=-999 required=5\r
+       tests=[DKIM_SIGNED=0.1, DKIM_VALID=-0.1, DKIM_VALID_AU=-0.1,\r
+       FREEMAIL_FROM=0.001, RCVD_IN_DNSWL_LOW=-0.7] autolearn=disabled\r
+Received: from olra.theworths.org ([127.0.0.1])\r
+       by localhost (olra.theworths.org [127.0.0.1]) (amavisd-new, port 10024)\r
+       with ESMTP id MuUKKbfYHxTC for <notmuch@notmuchmail.org>;\r
+       Tue,  4 Sep 2012 13:12:44 -0700 (PDT)\r
+Received: from mail-ey0-f181.google.com (mail-ey0-f181.google.com\r
+       [209.85.215.181]) (using TLSv1 with cipher RC4-SHA (128/128 bits))\r
+       (No client certificate requested)\r
+       by olra.theworths.org (Postfix) with ESMTPS id 28AD1431FAF\r
+       for <notmuch@notmuchmail.org>; Tue,  4 Sep 2012 13:12:44 -0700 (PDT)\r
+Received: by eaan10 with SMTP id n10so2473248eaa.26\r
+       for <notmuch@notmuchmail.org>; Tue, 04 Sep 2012 13:12:42 -0700 (PDT)\r
+DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=gmail.com; s=20120113;\r
+       h=from:to:subject:in-reply-to:references:user-agent:date:message-id\r
+       :mime-version:content-type:content-transfer-encoding;\r
+       bh=/0V4lbozO3w2BdidaTB3mWL6HxCXqSb8xE6rnngN3Q0=;\r
+       b=seUZ7gDgf8JS4Rv2bF3TOI/qaxc3yH5sX7Npn+QtNkQO3LjAIBMSpX32iRDz3kyLUG\r
+       +7DqAflJe5Tl49XYXjCaD2HO1jSRZV1gqAUpdQIBZ4OEdjORVuGeG5plgQhAXrelQgIw\r
+       D32eUB8NqR7jJdCz2YBcp5TK31fGx/z2aWkQGCkF9Miry0l+D5zt2sS7V3yNVSwvv0it\r
+       8seg7YW2pco+PoUwHIjZI1bAsu+IuHxaiqfRePOLrbF+PWw/YJTQIj1UX3hx7VyhC21M\r
+       G7ILw1MB9FviVkgPTVjFsfBxGpSn1naShaIO5fvTW3SSS23Lqq1QKbQHoB54j3/D4RGS\r
+       3GJw==\r
+Received: by 10.14.224.4 with SMTP id w4mr27959330eep.21.1346789562767;\r
+       Tue, 04 Sep 2012 13:12:42 -0700 (PDT)\r
+Received: from localhost ([2001:470:1f0b:14dd:224:d7ff:fee2:c588])\r
+       by mx.google.com with ESMTPS id u8sm48089016eel.11.2012.09.04.13.12.41\r
+       (version=TLSv1/SSLv3 cipher=OTHER);\r
+       Tue, 04 Sep 2012 13:12:41 -0700 (PDT)\r
+From: Dmitry Kurochkin <dmitry.kurochkin@gmail.com>\r
+To: Michal Nazarewicz <mina86@mina86.com>, notmuch@notmuchmail.org\r
+Subject: Re: [PATCH] Add notmuch-remove-duplicates.py script to contrib.\r
+In-Reply-To: <xa1tligpk1za.fsf@mina86.com>\r
+References: <1346784785-19746-1-git-send-email-dmitry.kurochkin@gmail.com>\r
+       <xa1tligpk1za.fsf@mina86.com>\r
+User-Agent: Notmuch/0.14+18~g79a73cd (http://notmuchmail.org) Emacs/23.4.1\r
+       (x86_64-pc-linux-gnu)\r
+Date: Wed, 05 Sep 2012 00:12:39 +0400\r
+Message-ID: <87d321sg20.fsf@gmail.com>\r
+MIME-Version: 1.0\r
+Content-Type: text/plain; charset=utf-8\r
+Content-Transfer-Encoding: quoted-printable\r
+X-BeenThere: notmuch@notmuchmail.org\r
+X-Mailman-Version: 2.1.13\r
+Precedence: list\r
+List-Id: "Use and development of the notmuch mail system."\r
+       <notmuch.notmuchmail.org>\r
+List-Unsubscribe: <http://notmuchmail.org/mailman/options/notmuch>,\r
+       <mailto:notmuch-request@notmuchmail.org?subject=unsubscribe>\r
+List-Archive: <http://notmuchmail.org/pipermail/notmuch>\r
+List-Post: <mailto:notmuch@notmuchmail.org>\r
+List-Help: <mailto:notmuch-request@notmuchmail.org?subject=help>\r
+List-Subscribe: <http://notmuchmail.org/mailman/listinfo/notmuch>,\r
+       <mailto:notmuch-request@notmuchmail.org?subject=subscribe>\r
+X-List-Received-Date: Tue, 04 Sep 2012 20:12:45 -0000\r
+\r
+Hi Michal.\r
+\r
+Michal Nazarewicz <mina86@mina86.com> writes:\r
+\r
+> On Tue, Sep 04 2012, Dmitry Kurochkin wrote:\r
+>> The script removes duplicate message files.  It takes no options.\r
+>>\r
+>> Files are assumed duplicates if their content is the same except for\r
+>> ignored headers.  Currently, the only ignored header is Received:.\r
+>> ---\r
+>>  contrib/notmuch-remove-duplicates.py |   95 +++++++++++++++++++++++++++=\r
++++++++\r
+>>  1 file changed, 95 insertions(+)\r
+>>  create mode 100755 contrib/notmuch-remove-duplicates.py\r
+>>\r
+>> diff --git a/contrib/notmuch-remove-duplicates.py b/contrib/notmuch-remo=\r
+ve-duplicates.py\r
+>> new file mode 100755\r
+>> index 0000000..dbe2e25\r
+>> --- /dev/null\r
+>> +++ b/contrib/notmuch-remove-duplicates.py\r
+>> @@ -0,0 +1,95 @@\r
+>> +#!/usr/bin/env python\r
+>> +\r
+>> +import sys\r
+>> +\r
+>> +IGNORED_HEADERS =3D [ "Received:" ]\r
+>> +\r
+>> +if len(sys.argv) !=3D 1:\r
+>> +    print "Usage: %s" % sys.argv[0]\r
+>> +    print\r
+>> +    print "The script removes duplicate message files.  Takes no option=\r
+s."\r
+>> +    print "Requires notmuch python module."\r
+>> +    print\r
+>> +    print "Files are assumed duplicates if their content is the same"\r
+>> +    print "except for the following headers: %s." % ", ".join(IGNORED_H=\r
+EADERS)\r
+>> +    exit(1)\r
+>\r
+> It's much better put inside a main() function, which is than called only\r
+> if the script is run directly.\r
+>\r
+\r
+Good point.  My python skill is pretty low :)\r
+\r
+>> +\r
+>> +import notmuch\r
+>> +import os\r
+>> +import time\r
+>> +\r
+>> +class MailComparator:\r
+>> +    """Checks if mail files are duplicates."""\r
+>> +    def __init__(self, filename):\r
+>> +        self.filename =3D filename\r
+>> +        self.mail =3D self.readFile(self.filename)\r
+>> +\r
+>> +    def isDuplicate(self, filename):\r
+>> +        return self.mail =3D=3D self.readFile(filename)\r
+>> +\r
+>> +    @staticmethod\r
+>> +    def readFile(filename):\r
+>> +        with open(filename) as f:\r
+>> +            data =3D ""\r
+>> +            while True:\r
+>> +                line =3D f.readline()\r
+>> +                for header in IGNORED_HEADERS:\r
+>> +                    if line.startswith(header):\r
+>\r
+> Case of headers should be ignored, but this does not ignore it.\r
+>\r
+\r
+It does.\r
+\r
+>> +                        # skip header continuation lines\r
+>> +                        while True:\r
+>> +                            line =3D f.readline()\r
+>> +                            if len(line) =3D=3D 0 or line[0] not in [" =\r
+", "\t"]:\r
+>> +                                break\r
+>> +                        break\r
+>\r
+> This will ignore line just after the ignored header.\r
+>\r
+\r
+The first header line is ignored as well because line is added to data\r
+in else block.\r
+\r
+>> +                else:\r
+>> +                    data +=3D line\r
+>> +                    if line =3D=3D "\n":\r
+>> +                        break\r
+>> +            data +=3D f.read()\r
+>> +            return data\r
+>> +\r
+>> +db =3D notmuch.Database()\r
+>> +query =3D db.create_query('*')\r
+>> +print "Number of messages: %s" % query.count_messages()\r
+>> +\r
+>> +files_count =3D 0\r
+>> +for root, dirs, files in os.walk(db.get_path()):\r
+>> +    if not root.startswith(os.path.join(db.get_path(), ".notmuch/")):\r
+>> +        files_count +=3D len(files)\r
+>> +print "Number of files: %s" % files_count\r
+>> +print "Estimated number of duplicates: %s" % (files_count - query.count=\r
+_messages())\r
+>> +\r
+>> +msgs =3D query.search_messages()\r
+>> +msg_count =3D 0\r
+>> +suspected_duplicates_count =3D 0\r
+>> +duplicates_count =3D 0\r
+>> +timestamp =3D time.time()\r
+>> +for msg in msgs:\r
+>> +    msg_count +=3D 1\r
+>> +    if len(msg.get_filenames()) > 1:\r
+>> +        filenames =3D msg.get_filenames()\r
+>> +        comparator =3D MailComparator(filenames.next())\r
+>> +        for filename in filenames:\r
+>\r
+> Strictly speaking, you need to compare each file to each file, and not\r
+> just every file to the first file.\r
+>\r
+>> +            if os.path.realpath(comparator.filename) =3D=3D os.path.rea=\r
+lpath(filename):\r
+>> +                print "Message '%s' has filenames pointing to the\r
+>> same file: '%s' '%s'" % (msg.get_message_id(), comparator.filename,\r
+>> filename)\r
+>\r
+> So why aren't those removed?\r
+>\r
+\r
+Because it is the same file indexed twice (probably because of\r
+symlinks).  We do not want to remove the only message file.\r
+\r
+>> +            elif comparator.isDuplicate(filename):\r
+>> +                os.remove(filename)\r
+>> +                duplicates_count +=3D 1\r
+>> +            else:\r
+>> +                #print "Potential duplicates: %s" % msg.get_message_id()\r
+>> +                suspected_duplicates_count +=3D 1\r
+>> +\r
+>> +    new_timestamp =3D time.time()\r
+>> +    if new_timestamp - timestamp > 1:\r
+>> +        timestamp =3D new_timestamp\r
+>> +        sys.stdout.write("\rProcessed %s messages, removed %s duplicate=\r
+s..." % (msg_count, duplicates_count))\r
+>> +        sys.stdout.flush()\r
+>> +\r
+>> +print "\rFinished. Processed %s messages, removed %s duplicates." % (ms=\r
+g_count, duplicates_count)\r
+>> +if duplicates_count > 0:\r
+>> +    print "You might want to run 'notmuch new' now."\r
+>> +\r
+>> +if suspected_duplicates_count > 0:\r
+>> +    print\r
+>> +    print "Found %s messages with duplicate IDs but different content."=\r
+ % suspected_duplicates_count\r
+>> +    print "Perhaps we should ignore more headers."\r
+>\r
+> Please consider the following instead (not tested):\r
+>\r
+\r
+Thanks for reviewing my poor python code :) I am afraid I do not have\r
+enough interest in improving it.  I just implemented a simple solution\r
+for my problem.  Though it looks like you already took time to rewrite\r
+the script.  Would be great if you send it as a proper patch obsoleting\r
+this one.\r
+\r
+Regards,\r
+  Dmitry\r
+\r
+>\r
+> #!/usr/bin/env python\r
+>\r
+> import collections\r
+> import notmuch\r
+> import os\r
+> import re\r
+> import sys\r
+> import time\r
+>\r
+>\r
+> IGNORED_HEADERS =3D [ 'Received' ]\r
+>\r
+>\r
+> isIgnoredHeadersLine =3D re.compile(\r
+>     r'^(?:%s)\s*:' % '|'.join(IGNORED_HEADERS),\r
+>     re.IGNORECASE).search\r
+>\r
+> doesStartWithWS =3D re.compile(r'^\s').search\r
+>\r
+>\r
+> def usage(argv0):\r
+>     print """Usage: %s [<query-string>]\r
+>\r
+> The script removes duplicate message files.  Takes no options."\r
+> Requires notmuch python module."\r
+>\r
+> Files are assumed duplicates if their content is the same"\r
+> except for the following headers: %s.""" % (argv0, ', '.join(IGNORED_HEAD=\r
+ERS))\r
+>\r
+>\r
+> def readMailFile(filename):\r
+>     with open(filename) as fd:\r
+>         data =3D []\r
+>         skip_header =3D False\r
+>         for line in fd:\r
+>             if doesStartWithWS(line):\r
+>                 if not skip_header:\r
+>                     data.append(line)\r
+>             elif isIgnoredHeadersLine(line):\r
+>                 skip_header =3D True\r
+>             else:\r
+>                 data.append(line)\r
+>                 if line =3D=3D '\n':\r
+>                     break\r
+>         data.append(fd.read())\r
+>         return ''.join(data)\r
+>\r
+>\r
+> def dedupMessage(msg):\r
+>     filenames =3D msg.get_filenames()\r
+>     if len(filenames) <=3D 1:\r
+>         return (0, 0)\r
+>\r
+>     realpaths =3D collections.defaultdict(list)\r
+>     contents =3D collections.defaultdict(list)\r
+>     for filename in filenames:\r
+>         real =3D os.path.realpath(filename)\r
+>         lst =3D realpaths[real]\r
+>         lst.append(filename)\r
+>         if len(lst) =3D=3D 1:\r
+>             contents[readMailFile(real)].append(real)\r
+>\r
+>     duplicates =3D 0\r
+>\r
+>     for filenames in contents.itervalues():\r
+>         if len(filenames) > 1:\r
+>             print 'Files with the same content:'\r
+>             print ' ', filenames.pop()\r
+>             duplicates +=3D len(filenames)\r
+>             for filename in filenames:\r
+>                 del realpaths[filename]\r
+>             #     os.remane(filename)\r
+>\r
+>     for real, filenames in realpaths.iteritems():\r
+>         if len(filenames) > 1:\r
+>             print 'Files pointing to the same message:'\r
+>             print ' ', filenames.pop()\r
+>             duplicates +=3D len(filenames)\r
+>             # for filename in filenames:\r
+>             #     os.remane(filename)\r
+>\r
+>     return (duplicates, len(realpaths) - 1)\r
+>\r
+>\r
+> def dedupQuery(query):\r
+>     print 'Number of messages: %s' % query.count_messages()\r
+>     msg_count =3D 0\r
+>     suspected_count =3D 0\r
+>     duplicates_count =3D 0\r
+>     timestamp =3D time.time()\r
+>     msgs =3D query.search_messages()\r
+>     for msg in msgs:\r
+>         msg_count +=3D 1\r
+>         d, s =3D dedupMessage(msg)\r
+>         duplicates_count +=3D d\r
+>         suspected_count +=3D d\r
+>\r
+>         new_timestamp =3D time.time()\r
+>         if new_timestamp - timestamp > 1:\r
+>             timestamp =3D new_timestamp\r
+>             sys.stdout.write('\rProcessed %s messages, removed %s duplica=\r
+tes...'\r
+>                              % (msg_count, duplicates_count))\r
+>             sys.stdout.flush()\r
+>\r
+>     print '\rFinished. Processed %s messages, removed %s duplicates.' % (\r
+>         msg_count, duplicates_count)\r
+>     if duplicates_count > 0:\r
+>         print 'You might want to run "notmuch new" now.'\r
+>\r
+>     if suspected_duplicates_count > 0:\r
+>         print """\r
+> Found %d messages with duplicate IDs but different content.\r
+> Perhaps we should ignore more headers.""" % suspected_count\r
+>\r
+>\r
+> def main(argv):\r
+>     if len(argv) =3D=3D 1:\r
+>         query =3D '*'\r
+>     elif len(argv) =3D=3D 2:\r
+>         query =3D argv[1]\r
+>     else:\r
+>         usage(argv[0])\r
+>         return 1\r
+>\r
+>     db =3D notmuch.Database()\r
+>     query =3D db.create_query(query)\r
+>     dedupQuery(db, query)\r
+>     return 0\r
+>\r
+>\r
+> if __name__ =3D=3D '__main__':\r
+>     sys.exit(main(sys.argv))\r
+>\r
+>\r
+>\r
+> --=20\r
+> Best regards,                                         _     _\r
+> .o. | Liege of Serenely Enlightened Majesty of      o' \,=3D./ `o\r
+> ..o | Computer Science,  Micha=C5=82 =E2=80=9Cmina86=E2=80=9D Nazarewicz =\r
+   (o o)\r
+> ooo +----<email/xmpp: mpn@google.com>--------------ooO--(_)--Ooo--\r