X-Git-Url: https://git.sommitrealweird.co.uk/rss2maildir.git/blobdiff_plain/629cdfb564c31d54236ba24cd6d891cba1c2308a..13a417ff459bfe827f45845f9ca5e04c08889e87:/rss2maildir.py?ds=sidebyside

diff --git a/rss2maildir.py b/rss2maildir.py
index 910b8a9..df7236a 100755
--- a/rss2maildir.py
+++ b/rss2maildir.py
@@ -48,14 +48,95 @@ from HTMLParser import HTMLParser
 
 class HTML2Text(HTMLParser):
     entities = {
-        u'amp': "&",
-        u'lt': "<",
-        u'gt': ">",
-        u'pound': "Â£",
-        u'copy': "Â©",
-        u'apos': "'",
-        u'quot': "\"",
-        u'nbsp': " ",
+        u'amp': u'&',
+        u'lt': u'<',
+        u'gt': u'>',
+        u'pound': u'Â£',
+        u'copy': u'Â©',
+        u'apos': u'\'',
+        u'quot': u'"',
+        u'nbsp': u' ',
+        u'ldquo': u'â',
+        u'rdquo': u'â',
+        u'lsquo': u'â',
+        u'rsquo': u'â',
+        u'laquo': u'Â«',
+        u'raquo': u'Â»',
+        u'lsaquo': u'â¹',
+        u'rsaquo': u'âº',
+        u'bull': u'â¢',
+        u'middot': u'Â·',
+        u'deg': u'Â°',
+        u'helip': u'â¦',
+        u'trade': u'â¢',
+        u'reg': u'Â®',
+        u'agrave': u'Ã ',
+        u'Agrave': u'Ã',
+        u'egrave': u'Ã¨',
+        u'Egrave': u'Ã',
+        u'igrave': u'Ã¬',
+        u'Igrave': u'Ã',
+        u'ograve': u'Ã²',
+        u'Ograve': u'Ã',
+        u'ugrave': u'Ã¹',
+        u'Ugrave': u'Ã',
+        u'aacute': u'Ã¡',
+        u'Aacute': u'Ã',
+        u'eacute': u'Ã©',
+        u'Eacute': u'Ã',
+        u'iacute': u'Ã­',
+        u'Iacute': u'Ã',
+        u'oacute': u'Ã³',
+        u'Oacute': u'Ã',
+        u'uacute': u'Ãº',
+        u'Uacute': u'Ã',
+        u'yactue': u'Ã½',
+        u'Yacute': u'Ã',
+        u'acirc': u'Ã¢',
+        u'Acirc': u'Ã',
+        u'ecirc': u'Ãª',
+        u'Ecirc': u'Ã',
+        u'icirc': u'Ã®',
+        u'Icirc': u'Ã',
+        u'ocirc': u'Ã´',
+        u'Ocirc': u'Ã',
+        u'ucirc': u'Ã»',
+        u'Ucirc': u'Ã',
+        u'atilde': u'Ã£',
+        u'Atilde': u'Ã',
+        u'ntilde': u'Ã±',
+        u'Ntilde': u'Ã',
+        u'otilde': u'Ãµ',
+        u'Otilde': u'Ã',
+        u'auml': u'Ã¤',
+        u'Auml': u'Ã',
+        u'euml': u'Ã«',
+        u'Euml': u'Ã',
+        u'iuml': u'Ã¯',
+        u'Iuml': u'Ã',
+        u'ouml': u'Ã¶',
+        u'Ouml': u'Ã',
+        u'uuml': u'Ã¼',
+        u'Uuml': u'Ã',
+        u'yuml': u'Ã¿',
+        u'Yuml': u'Å¸',
+        u'iexcl': u'Â¡',
+        u'iquest': u'Â¿',
+        u'ccedil': u'Ã§',
+        u'Ccedil': u'Ã',
+        u'oelig': u'Å',
+        u'OElig': u'Å',
+        u'szlig': u'Ã',
+        u'oslash': u'Ã¸',
+        u'Oslash': u'Ã',
+        u'aring': u'Ã¥',
+        u'Aring': u'Ã',
+        u'aelig': u'Ã¦',
+        u'AElig': u'Ã',
+        u'thorn': u'Ã¾',
+        u'THORN': u'Ã',
+        u'eth': u'Ã°',
+        u'ETH': u'Ã',
         }
 
     blockleveltags = [
@@ -70,6 +151,9 @@ class HTML2Text(HTMLParser):
         u'ul',
         u'ol',
         u'dl',
+        u'li',
+        u'dt',
+        u'dd',
         u'div',
         #u'blockquote',
         ]
@@ -96,6 +180,7 @@ class HTML2Text(HTMLParser):
         self.ignorenodata = False
         self.listcount = []
         self.urls = []
+        self.images = {}
         HTMLParser.__init__(self)
 
     def handle_starttag(self, tag, attrs):
@@ -148,7 +233,7 @@ class HTML2Text(HTMLParser):
             elif tag_name == u'a':
                 for attr in attrs:
                     if attr[0].lower() == u'href':
-                        self.urls.append(attr[1])
+                        self.urls.append(attr[1].decode('utf-8'))
                 self.curdata = self.curdata + u'`'
                 self.opentags.append(tag_name)
                 return
@@ -180,24 +265,35 @@ class HTML2Text(HTMLParser):
         url = u''
         for attr in attrs:
             if attr[0] == 'alt':
-                alt = attr[1]
+                alt = attr[1].decode('utf-8')
             elif attr[0] == 'src':
-                url = attr[1]
+                url = attr[1].decode('utf-8')
         if url:
-            self.curdata = self.curdata \
-                + u' [img:' \
-                + unicode( \
-                    url.encode('utf-8'), \
-                    'utf-8')
             if alt:
-                self.curdata = self.curdata \
-                    + u'(' \
-                    + unicode( \
-                        alt.encode('utf-8'), \
-                        'utf-8') \
-                    + u')'
-            self.curdata = self.curdata \
-                + u']'
+                if self.images.has_key(alt):
+                    if self.images[alt]["url"] == url:
+                        self.curdata = self.curdata \
+                            + u'|%s|' %(alt,)
+                    else:
+                        while self.images.has_key(alt):
+                            alt = alt + "_"
+                        self.images[alt]["url"] = url
+                        self.curdata = self.curdata \
+                            + u'|%s|' %(alt,)
+                else:
+                    self.images[alt] = {}
+                    self.images[alt]["url"] = url
+                    self.curdata = self.curdata \
+                        + u'|%s|' %(alt,)
+            else:
+                if self.images.has_key(url):
+                    self.curdata = self.curdata \
+                        + u'|%s|' %(url,)
+                else:
+                    self.images[url] = {}
+                    self.images[url]["url"] =url
+                    self.curdata = self.curdata \
+                        + u'|%s|' %(url,)
 
     def handle_curdata(self):
 
@@ -223,17 +319,20 @@ class HTML2Text(HTMLParser):
             if self.ignorenodata:
                 newlinerequired = False
             self.ignorenodata = False
-            if newlinerequired \
-                and len(self.text) > 2 \
-                and self.text[-1] != u'\n' \
-                and self.text[-2] != u'\n':
+            if newlinerequired:
+                if tag_thats_done in [u'dt', u'dd', u'li'] \
+                    and len(self.text) > 1 \
+                    and self.text[-1] != u'\n':
+                        self.text = self.text + u'\n'
+                elif len(self.text) > 2 \
+                    and self.text[-1] != u'\n' \
+                    and self.text[-2] != u'\n':
                     self.text = self.text + u'\n\n'
 
         if tag_thats_done in ["h1", "h2", "h3", "h4", "h5", "h6"]:
             underline = u''
             underlinechar = u'='
-            headingtext = unicode( \
-                self.curdata.encode("utf-8").strip(), "utf-8")
+            headingtext = " ".join(self.curdata.split())
             seperator = u'\n' + u' '*self.indentlevel
             headingtext = seperator.join( \
                 textwrap.wrap( \
@@ -254,11 +353,12 @@ class HTML2Text(HTMLParser):
                 underline = u' ' * self.indentlevel \
                     + underlinechar * len(headingtext)
             self.text = self.text \
-                + headingtext.encode("utf-8") + u'\n' \
+                + headingtext + u'\n' \
                 + underline
         elif tag_thats_done in [u'p', u'div']:
             paragraph = unicode( \
-                self.curdata.strip().encode("utf-8"), "utf-8")
+                " ".join(self.curdata.strip().encode("utf-8").split()), \
+                "utf-8")
             seperator = u'\n' + u' ' * self.indentlevel
             self.text = self.text \
                 + u' ' * self.indentlevel \
@@ -270,7 +370,8 @@ class HTML2Text(HTMLParser):
                 self.curdata.encode("utf-8"), "utf-8")
         elif tag_thats_done == u'blockquote':
             quote = unicode( \
-                self.curdata.encode("utf-8").strip(), "utf-8")
+                " ".join(self.curdata.encode("utf-8").strip().split()), \
+                "utf-8")
             seperator = u'\n' + u' ' * self.indentlevel + u'> '
             if len(self.text) > 0 and self.text[-1] != u'\n':
                 self.text = self.text + u'\n'
@@ -322,7 +423,9 @@ class HTML2Text(HTMLParser):
                 )
             self.curdata = u''
         elif tag_thats_done == u'dt':
-            definition = unicode(self.curdata.encode("utf-8").strip(), "utf-8")
+            definition = unicode(" ".join( \
+                    self.curdata.encode("utf-8").strip().split()), \
+                "utf-8")
             if len(self.text) > 0 and self.text[-1] != u'\n':
                 self.text = self.text + u'\n\n'
             elif len(self.text) > 1 and self.text[-2] != u'\n':
@@ -335,7 +438,9 @@ class HTML2Text(HTMLParser):
                         self.textwidth - self.indentlevel - 1))
             self.curdata = u''
         elif tag_thats_done == u'dd':
-            definition = unicode(self.curdata.encode("utf-8").strip(), "utf-8")
+            definition = unicode(" ".join( \
+                    self.curdata.encode("utf-8").strip().split()),
+                "utf-8")
             if len(definition) > 0:
                 if len(self.text) > 0 and self.text[-1] != u'\n':
                     self.text = self.text + u'\n'
@@ -411,7 +516,7 @@ class HTML2Text(HTMLParser):
     def handle_data(self, data):
         if len(self.opentags) == 0:
             self.opentags.append(u'p')
-        self.curdata = self.curdata + unicode(data, "utf-8")
+        self.curdata = self.curdata + data.decode("utf-8")
 
     def handle_entityref(self, name):
         entity = name
@@ -436,6 +541,12 @@ class HTML2Text(HTMLParser):
         if len(self.urls) > 0:
             self.text = self.text + u'\n__ ' + u'\n__ '.join(self.urls) + u'\n'
             self.urls = []
+        if len(self.images.keys()) > 0:
+            self.text = self.text + u'\n.. ' \
+                + u'.. '.join( \
+                    ["|%s| image:: %s" %(a, self.images[a]["url"]) \
+                for a in self.images.keys()]) + u'\n'
+            self.images = {}
         return self.text
 
 def open_url(method, url):
@@ -721,11 +832,13 @@ if __name__ == "__main__":
     elif scp.has_option("general", "state_dir"):
         new_state_dir = scp.get("general", "state_dir")
         try:
-            mode = os.stat(state_dir)[stat.ST_MODE]
+            mode = os.stat(new_state_dir)[stat.ST_MODE]
             if not stat.S_ISDIR(mode):
                 sys.stderr.write( \
                     "State directory (%s) is not a directory\n" %(state_dir))
                 sys.exit(1)
+            else:
+                state_dir = new_state_dir
         except:
             # try to create it
             try: