Normalise spaces where they should be.

[rss2maildir.git] / rss2maildir.py
diff --git a/rss2maildir.py b/rss2maildir.py

index 113d931a1db7bf162f603ce1cc646137a1f97b9c..ce6c3426b54fff080b6639a09e85269375fb170a 100755 (executable)
--- a/rss2maildir.py
+++ b/rss2maildir.py
@@ -70,6 +70,9 @@ class HTML2Text(HTMLParser):
          u'ul',
          u'ol',
          u'dl',
+        u'li',
+        u'dt',
+        u'dd',
          u'div',
          #u'blockquote',
          ]
@@ -148,7 +151,7 @@ class HTML2Text(HTMLParser):
              elif tag_name == u'a':
                  for attr in attrs:
                      if attr[0].lower() == u'href':
-                        self.urls.append(attr[1])
+                        self.urls.append(attr[1].decode('utf-8'))
                  self.curdata = self.curdata + u'`'
                  self.opentags.append(tag_name)
                  return
@@ -219,17 +222,20 @@ class HTML2Text(HTMLParser):
              if self.ignorenodata:
                  newlinerequired = False
              self.ignorenodata = False
-            if newlinerequired \
-                and len(self.text) > 2 \
-                and self.text[-1] != u'\n' \
-                and self.text[-2] != u'\n':
+            if newlinerequired:
+                if tag_thats_done in [u'dt', u'dd', u'li'] \
+                    and len(self.text) > 1 \
+                    and self.text[-1] != u'\n':
+                        self.text = self.text + u'\n'
+                elif len(self.text) > 2 \
+                    and self.text[-1] != u'\n' \
+                    and self.text[-2] != u'\n':
                      self.text = self.text + u'\n\n'
  
          if tag_thats_done in ["h1", "h2", "h3", "h4", "h5", "h6"]:
              underline = u''
              underlinechar = u'='
-            headingtext = unicode( \
-                self.curdata.encode("utf-8").strip(), "utf-8")
+            headingtext = " ".join(self.curdata.split())
              seperator = u'\n' + u' '*self.indentlevel
              headingtext = seperator.join( \
                  textwrap.wrap( \
@@ -250,11 +256,12 @@ class HTML2Text(HTMLParser):
                  underline = u' ' * self.indentlevel \
                      + underlinechar * len(headingtext)
              self.text = self.text \
-                + headingtext.encode("utf-8") + u'\n' \
+                + headingtext + u'\n' \
                  + underline
          elif tag_thats_done in [u'p', u'div']:
              paragraph = unicode( \
-                self.curdata.strip().encode("utf-8"), "utf-8")
+                " ".join(self.curdata.strip().encode("utf-8").split()), \
+                "utf-8")
              seperator = u'\n' + u' ' * self.indentlevel
              self.text = self.text \
                  + u' ' * self.indentlevel \
@@ -263,10 +270,11 @@ class HTML2Text(HTMLParser):
                          paragraph, self.textwidth - self.indentlevel))
          elif tag_thats_done == "pre":
              self.text = self.text + unicode( \
-                self.curdata.encode("utf-8"), "utf-8")
+                " ".join(self.curdata.encode("utf-8").split()), "utf-8")
          elif tag_thats_done == u'blockquote':
              quote = unicode( \
-                self.curdata.encode("utf-8").strip(), "utf-8")
+                " ".join(self.curdata.encode("utf-8").strip().split()), \
+                "utf-8")
              seperator = u'\n' + u' ' * self.indentlevel + u'> '
              if len(self.text) > 0 and self.text[-1] != u'\n':
                  self.text = self.text + u'\n'
@@ -318,7 +326,9 @@ class HTML2Text(HTMLParser):
                  )
              self.curdata = u''
          elif tag_thats_done == u'dt':
-            definition = unicode(self.curdata.encode("utf-8").strip(), "utf-8")
+            definition = unicode(" ".join( \
+                    self.curdata.encode("utf-8").strip().split()), \
+                "utf-8")
              if len(self.text) > 0 and self.text[-1] != u'\n':
                  self.text = self.text + u'\n\n'
              elif len(self.text) > 1 and self.text[-2] != u'\n':
@@ -331,7 +341,9 @@ class HTML2Text(HTMLParser):
                          self.textwidth - self.indentlevel - 1))
              self.curdata = u''
          elif tag_thats_done == u'dd':
-            definition = unicode(self.curdata.encode("utf-8").strip(), "utf-8")
+            definition = unicode(" ".join( \
+                    self.curdata.encode("utf-8").strip().split()),
+                "utf-8")
              if len(definition) > 0:
                  if len(self.text) > 0 and self.text[-1] != u'\n':
                      self.text = self.text + u'\n'
@@ -407,7 +419,7 @@ class HTML2Text(HTMLParser):
      def handle_data(self, data):
          if len(self.opentags) == 0:
              self.opentags.append(u'p')
-        self.curdata = self.curdata + unicode(data, "utf-8")
+        self.curdata = self.curdata + data.decode("utf-8")
  
      def handle_entityref(self, name):
          entity = name
@@ -717,11 +729,13 @@ if __name__ == "__main__":
      elif scp.has_option("general", "state_dir"):
          new_state_dir = scp.get("general", "state_dir")
          try:
-            mode = os.stat(state_dir)[stat.ST_MODE]
+            mode = os.stat(new_state_dir)[stat.ST_MODE]
              if not stat.S_ISDIR(mode):
                  sys.stderr.write( \
                      "State directory (%s) is not a directory\n" %(state_dir))
                  sys.exit(1)
+            else:
+                state_dir = new_state_dir
          except:
              # try to create it
              try: