[PATCH] Add notmuch-remove-duplicates.py script to contrib.
authorDmitry Kurochkin <dmitry.kurochkin@gmail.com>
Tue, 4 Sep 2012 18:53:05 +0000 (22:53 +0400)
committerW. Trevor King <wking@tremily.us>
Fri, 7 Nov 2014 17:49:22 +0000 (09:49 -0800)
48/0c467ec54a5db31a1f5cf7ba80c3bdbb93745f [new file with mode: 0644]

diff --git a/48/0c467ec54a5db31a1f5cf7ba80c3bdbb93745f b/48/0c467ec54a5db31a1f5cf7ba80c3bdbb93745f
new file mode 100644 (file)
index 0000000..02a87e9
--- /dev/null
@@ -0,0 +1,172 @@
+Return-Path: <dmitry.kurochkin@gmail.com>\r
+X-Original-To: notmuch@notmuchmail.org\r
+Delivered-To: notmuch@notmuchmail.org\r
+Received: from localhost (localhost [127.0.0.1])\r
+       by olra.theworths.org (Postfix) with ESMTP id 33790431FB6\r
+       for <notmuch@notmuchmail.org>; Tue,  4 Sep 2012 11:53:12 -0700 (PDT)\r
+X-Virus-Scanned: Debian amavisd-new at olra.theworths.org\r
+X-Spam-Flag: NO\r
+X-Spam-Score: -0.799\r
+X-Spam-Level: \r
+X-Spam-Status: No, score=-0.799 tagged_above=-999 required=5\r
+       tests=[DKIM_SIGNED=0.1, DKIM_VALID=-0.1, DKIM_VALID_AU=-0.1,\r
+       FREEMAIL_FROM=0.001, RCVD_IN_DNSWL_LOW=-0.7] autolearn=disabled\r
+Received: from olra.theworths.org ([127.0.0.1])\r
+       by localhost (olra.theworths.org [127.0.0.1]) (amavisd-new, port 10024)\r
+       with ESMTP id A2Pzhqsv8xA6 for <notmuch@notmuchmail.org>;\r
+       Tue,  4 Sep 2012 11:53:11 -0700 (PDT)\r
+Received: from mail-ee0-f53.google.com (mail-ee0-f53.google.com\r
+ [74.125.83.53])       (using TLSv1 with cipher RC4-SHA (128/128 bits))        (No client\r
+ certificate requested)        by olra.theworths.org (Postfix) with ESMTPS id\r
+ 4B2D0431FAF   for <notmuch@notmuchmail.org>; Tue,  4 Sep 2012 11:53:11 -0700\r
+ (PDT)\r
+Received: by eekb47 with SMTP id b47so2947704eek.26\r
+       for <notmuch@notmuchmail.org>; Tue, 04 Sep 2012 11:53:08 -0700 (PDT)\r
+DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=gmail.com; s=20120113;\r
+       h=from:to:subject:date:message-id:x-mailer;\r
+       bh=dozgH3mWp36ddi1pgxlbLiGB1CaZxhWjD8Nva5iDvAc=;\r
+       b=mA8A+Gnzud/GFb9xDcnbG27IpaeEWC4DvYRj9DEIZbpASSVp4Medlim0oZbjXHM37f\r
+       nmFHfPmz26nHqklfz1rR8OyxM9ssCZGaCxKAqaQdWYuKZZfFwo3TKIWDOUWzeFxI2u2N\r
+       gEH5pIC2G1LyihObX9omjUr+E/AtsWcgW+Pvk44Mv1eM6AykJtCrxql0JhOGzV4LrJq0\r
+       X2Ik17BOtWq6AXtsDGNB9ZEbdT7m2Ft+3JmIiBimQ5VsYhi3GFARroq8a4LtOBWFAzK8\r
+       z1qF7b7N+sPeVsue23xzVnQfm/hptZVFhbT2o4JNf3Y3vCgLazvLPl9ZU9lZ1FMdeOtK\r
+       y+Fg==\r
+Received: by 10.14.204.72 with SMTP id g48mr27528941eeo.45.1346784788676;\r
+       Tue, 04 Sep 2012 11:53:08 -0700 (PDT)\r
+Received: from localhost ([2001:470:1f0b:14dd:224:d7ff:fee2:c588])\r
+       by mx.google.com with ESMTPS id e7sm47648765eep.2.2012.09.04.11.53.07\r
+       (version=TLSv1/SSLv3 cipher=OTHER);\r
+       Tue, 04 Sep 2012 11:53:07 -0700 (PDT)\r
+From: Dmitry Kurochkin <dmitry.kurochkin@gmail.com>\r
+To: notmuch@notmuchmail.org\r
+Subject: [PATCH] Add notmuch-remove-duplicates.py script to contrib.\r
+Date: Tue,  4 Sep 2012 22:53:05 +0400\r
+Message-Id: <1346784785-19746-1-git-send-email-dmitry.kurochkin@gmail.com>\r
+X-Mailer: git-send-email 1.7.10.4\r
+X-BeenThere: notmuch@notmuchmail.org\r
+X-Mailman-Version: 2.1.13\r
+Precedence: list\r
+List-Id: "Use and development of the notmuch mail system."\r
+       <notmuch.notmuchmail.org>\r
+List-Unsubscribe: <http://notmuchmail.org/mailman/options/notmuch>,\r
+       <mailto:notmuch-request@notmuchmail.org?subject=unsubscribe>\r
+List-Archive: <http://notmuchmail.org/pipermail/notmuch>\r
+List-Post: <mailto:notmuch@notmuchmail.org>\r
+List-Help: <mailto:notmuch-request@notmuchmail.org?subject=help>\r
+List-Subscribe: <http://notmuchmail.org/mailman/listinfo/notmuch>,\r
+       <mailto:notmuch-request@notmuchmail.org?subject=subscribe>\r
+X-List-Received-Date: Tue, 04 Sep 2012 18:53:12 -0000\r
+\r
+The script removes duplicate message files.  It takes no options.\r
+\r
+Files are assumed duplicates if their content is the same except for\r
+ignored headers.  Currently, the only ignored header is Received:.\r
+---\r
+ contrib/notmuch-remove-duplicates.py |   95 ++++++++++++++++++++++++++++++++++\r
+ 1 file changed, 95 insertions(+)\r
+ create mode 100755 contrib/notmuch-remove-duplicates.py\r
+\r
+diff --git a/contrib/notmuch-remove-duplicates.py b/contrib/notmuch-remove-duplicates.py\r
+new file mode 100755\r
+index 0000000..dbe2e25\r
+--- /dev/null\r
++++ b/contrib/notmuch-remove-duplicates.py\r
+@@ -0,0 +1,95 @@\r
++#!/usr/bin/env python\r
++\r
++import sys\r
++\r
++IGNORED_HEADERS = [ "Received:" ]\r
++\r
++if len(sys.argv) != 1:\r
++    print "Usage: %s" % sys.argv[0]\r
++    print\r
++    print "The script removes duplicate message files.  Takes no options."\r
++    print "Requires notmuch python module."\r
++    print\r
++    print "Files are assumed duplicates if their content is the same"\r
++    print "except for the following headers: %s." % ", ".join(IGNORED_HEADERS)\r
++    exit(1)\r
++\r
++import notmuch\r
++import os\r
++import time\r
++\r
++class MailComparator:\r
++    """Checks if mail files are duplicates."""\r
++    def __init__(self, filename):\r
++        self.filename = filename\r
++        self.mail = self.readFile(self.filename)\r
++\r
++    def isDuplicate(self, filename):\r
++        return self.mail == self.readFile(filename)\r
++\r
++    @staticmethod\r
++    def readFile(filename):\r
++        with open(filename) as f:\r
++            data = ""\r
++            while True:\r
++                line = f.readline()\r
++                for header in IGNORED_HEADERS:\r
++                    if line.startswith(header):\r
++                        # skip header continuation lines\r
++                        while True:\r
++                            line = f.readline()\r
++                            if len(line) == 0 or line[0] not in [" ", "\t"]:\r
++                                break\r
++                        break\r
++                else:\r
++                    data += line\r
++                    if line == "\n":\r
++                        break\r
++            data += f.read()\r
++            return data\r
++\r
++db = notmuch.Database()\r
++query = db.create_query('*')\r
++print "Number of messages: %s" % query.count_messages()\r
++\r
++files_count = 0\r
++for root, dirs, files in os.walk(db.get_path()):\r
++    if not root.startswith(os.path.join(db.get_path(), ".notmuch/")):\r
++        files_count += len(files)\r
++print "Number of files: %s" % files_count\r
++print "Estimated number of duplicates: %s" % (files_count - query.count_messages())\r
++\r
++msgs = query.search_messages()\r
++msg_count = 0\r
++suspected_duplicates_count = 0\r
++duplicates_count = 0\r
++timestamp = time.time()\r
++for msg in msgs:\r
++    msg_count += 1\r
++    if len(msg.get_filenames()) > 1:\r
++        filenames = msg.get_filenames()\r
++        comparator = MailComparator(filenames.next())\r
++        for filename in filenames:\r
++            if os.path.realpath(comparator.filename) == os.path.realpath(filename):\r
++                print "Message '%s' has filenames pointing to the same file: '%s' '%s'" % (msg.get_message_id(), comparator.filename, filename)\r
++            elif comparator.isDuplicate(filename):\r
++                os.remove(filename)\r
++                duplicates_count += 1\r
++            else:\r
++                #print "Potential duplicates: %s" % msg.get_message_id()\r
++                suspected_duplicates_count += 1\r
++\r
++    new_timestamp = time.time()\r
++    if new_timestamp - timestamp > 1:\r
++        timestamp = new_timestamp\r
++        sys.stdout.write("\rProcessed %s messages, removed %s duplicates..." % (msg_count, duplicates_count))\r
++        sys.stdout.flush()\r
++\r
++print "\rFinished. Processed %s messages, removed %s duplicates." % (msg_count, duplicates_count)\r
++if duplicates_count > 0:\r
++    print "You might want to run 'notmuch new' now."\r
++\r
++if suspected_duplicates_count > 0:\r
++    print\r
++    print "Found %s messages with duplicate IDs but different content." % suspected_duplicates_count\r
++    print "Perhaps we should ignore more headers."\r
+-- \r
+1.7.10.4\r
+\r