Normalise spaces where they should be.

author Brett Parker <iDunno@sommitrealweird.co.uk>

Sat, 1 Mar 2008 20:57:10 +0000 (20:57 +0000)

committer Brett Parker <iDunno@sommitrealweird.co.uk>

Sat, 1 Mar 2008 20:57:10 +0000 (20:57 +0000)
author Brett Parker <iDunno@sommitrealweird.co.uk>
Sat, 1 Mar 2008 20:57:10 +0000 (20:57 +0000)
committer Brett Parker <iDunno@sommitrealweird.co.uk>
Sat, 1 Mar 2008 20:57:10 +0000 (20:57 +0000)
diff --git a/rss2maildir.py b/rss2maildir.py

index 2af32bc9105a49201e2bbba49d87b5e2cbcda1da..ce6c3426b54fff080b6639a09e85269375fb170a 100755 (executable)
--- a/rss2maildir.py
+++ b/rss2maildir.py
@@ -235,7 +235,7 @@ class HTML2Text(HTMLParser):
          if tag_thats_done in ["h1", "h2", "h3", "h4", "h5", "h6"]:
              underline = u''
              underlinechar = u'='
          if tag_thats_done in ["h1", "h2", "h3", "h4", "h5", "h6"]:
              underline = u''
              underlinechar = u'='
-            headingtext = self.curdata
+            headingtext = " ".join(self.curdata.split())
              seperator = u'\n' + u' '*self.indentlevel
              headingtext = seperator.join( \
                  textwrap.wrap( \
              seperator = u'\n' + u' '*self.indentlevel
              headingtext = seperator.join( \
                  textwrap.wrap( \
@@ -260,7 +260,8 @@ class HTML2Text(HTMLParser):
                  + underline
          elif tag_thats_done in [u'p', u'div']:
              paragraph = unicode( \
                  + underline
          elif tag_thats_done in [u'p', u'div']:
              paragraph = unicode( \
-                self.curdata.strip().encode("utf-8"), "utf-8")
+                " ".join(self.curdata.strip().encode("utf-8").split()), \
+                "utf-8")
              seperator = u'\n' + u' ' * self.indentlevel
              self.text = self.text \
                  + u' ' * self.indentlevel \
              seperator = u'\n' + u' ' * self.indentlevel
              self.text = self.text \
                  + u' ' * self.indentlevel \
@@ -269,10 +270,11 @@ class HTML2Text(HTMLParser):
                          paragraph, self.textwidth - self.indentlevel))
          elif tag_thats_done == "pre":
              self.text = self.text + unicode( \
                          paragraph, self.textwidth - self.indentlevel))
          elif tag_thats_done == "pre":
              self.text = self.text + unicode( \
-                self.curdata.encode("utf-8"), "utf-8")
+                " ".join(self.curdata.encode("utf-8").split()), "utf-8")
          elif tag_thats_done == u'blockquote':
              quote = unicode( \
          elif tag_thats_done == u'blockquote':
              quote = unicode( \
-                self.curdata.encode("utf-8").strip(), "utf-8")
+                " ".join(self.curdata.encode("utf-8").strip().split()), \
+                "utf-8")
              seperator = u'\n' + u' ' * self.indentlevel + u'> '
              if len(self.text) > 0 and self.text[-1] != u'\n':
                  self.text = self.text + u'\n'
              seperator = u'\n' + u' ' * self.indentlevel + u'> '
              if len(self.text) > 0 and self.text[-1] != u'\n':
                  self.text = self.text + u'\n'
@@ -324,7 +326,9 @@ class HTML2Text(HTMLParser):
                  )
              self.curdata = u''
          elif tag_thats_done == u'dt':
                  )
              self.curdata = u''
          elif tag_thats_done == u'dt':
-            definition = unicode(self.curdata.encode("utf-8").strip(), "utf-8")
+            definition = unicode(" ".join( \
+                    self.curdata.encode("utf-8").strip().split()), \
+                "utf-8")
              if len(self.text) > 0 and self.text[-1] != u'\n':
                  self.text = self.text + u'\n\n'
              elif len(self.text) > 1 and self.text[-2] != u'\n':
              if len(self.text) > 0 and self.text[-1] != u'\n':
                  self.text = self.text + u'\n\n'
              elif len(self.text) > 1 and self.text[-2] != u'\n':
@@ -337,7 +341,9 @@ class HTML2Text(HTMLParser):
                          self.textwidth - self.indentlevel - 1))
              self.curdata = u''
          elif tag_thats_done == u'dd':
                          self.textwidth - self.indentlevel - 1))
              self.curdata = u''
          elif tag_thats_done == u'dd':
-            definition = unicode(self.curdata.encode("utf-8").strip(), "utf-8")
+            definition = unicode(" ".join( \
+                    self.curdata.encode("utf-8").strip().split()),
+                "utf-8")
              if len(definition) > 0:
                  if len(self.text) > 0 and self.text[-1] != u'\n':
                      self.text = self.text + u'\n'
              if len(definition) > 0:
                  if len(self.text) > 0 and self.text[-1] != u'\n':
                      self.text = self.text + u'\n'
diff --git a/tests/expected/non-normalised-spacing.txt b/tests/expected/non-normalised-spacing.txt

new file mode 100644 (file)

index 0000000..50d8fef
--- /dev/null
+++ b/tests/expected/non-normalised-spacing.txt
@@ -0,0 +1,4 @@
+This has some odd spacing
+=========================
+
+It's really great and hopefully shouldn't be too bad over all
diff --git a/tests/html/non-normalised-spacing.html b/tests/html/non-normalised-spacing.html

new file mode 100644 (file)

index 0000000..0c85190
--- /dev/null
+++ b/tests/html/non-normalised-spacing.html
@@ -0,0 +1,7 @@
+<h1>This  has some     odd    spacing</h1>
+<p>It's really  great    and hopefully  shouldn't
+be
+
+too bad over
+all
+</p>
diff --git a/tests/unittests/SpacingTests.py b/tests/unittests/SpacingTests.py

new file mode 100755 (executable)

index 0000000..849e6ba
--- /dev/null
+++ b/tests/unittests/SpacingTests.py
@@ -0,0 +1,18 @@
+#!/usr/bin/python
+
+import unittest
+import os
+
+import ParsingTests
+
+class SpacingTests(ParsingTests.ParsingTest):
+    def testNormalisingSpacing(self):
+        return self.runParsingTest("non-normalised-spacing")
+
+def suite():
+    suite = unittest.TestSuite()
+    suite.addTest(SpacingTests("testNormalisingSpacing"))
+    return suite
+
+if __name__ == "__main__":
+    unittest.main()
author	Brett Parker <iDunno@sommitrealweird.co.uk>
	Sat, 1 Mar 2008 20:57:10 +0000 (20:57 +0000)
committer	Brett Parker <iDunno@sommitrealweird.co.uk>
	Sat, 1 Mar 2008 20:57:10 +0000 (20:57 +0000)
rss2maildir.py		patch \| blob \| history
tests/expected/non-normalised-spacing.txt	[new file with mode: 0644]	patch \| blob
tests/html/non-normalised-spacing.html	[new file with mode: 0644]	patch \| blob
tests/unittests/SpacingTests.py	[new file with mode: 0755]	patch \| blob