Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
29 changes: 29 additions & 0 deletions plugin.py
Original file line number Diff line number Diff line change
Expand Up @@ -72,6 +72,21 @@ def clean(self, msg):
return re.sub(r'\s+', ' ', cleaned)


def charset_from_contenttype(self, contenttype):
"""Return the charset declared in a Content-Type header, or None.

e.g. 'text/html; charset=utf-8' -> 'utf-8', 'text/html' -> None.
"""
if not contenttype:
return None
parts = contenttype.split('charset=')
if len(parts) == 2:
# Trim anything after the charset (further parameters) and any
# surrounding quotes/whitespace.
return parts[1].split(';')[0].strip().strip('"\'') or None
return None


def doPrivmsg(self, irc, msg):
if ircmsgs.isCtcp(msg) and not ircmsgs.isAction(msg):
return
Expand Down Expand Up @@ -106,6 +121,20 @@ def fetch_url(self, irc, channel, url):
return
html = response.read(MAXSIZE)
contenttype = response.info().getheader('Content-Type')
# Decode the response body using the charset declared in the
# Content-Type header *before* handing it to the (byte oriented)
# HTML parsers. Without this lxml assumes latin-1 and mangles
# non-ascii page titles, e.g. "GitHub · Social Coding" instead
# of "GitHub \xb7 Social Coding" (issue #2, a regression from the
# switch to lxml).
charset = self.charset_from_contenttype(contenttype)
if charset:
try:
html = html.decode(charset, 'replace')
except (LookupError, ValueError):
# Unknown/garbage charset: leave the bytes untouched and
# let the downstream parsers do their best.
pass
if 'text/html' in contenttype:
# Scrub potentially malformed HTML with lxml first
try:
Expand Down
20 changes: 20 additions & 0 deletions test.py
Original file line number Diff line number Diff line change
Expand Up @@ -32,5 +32,25 @@
class DetrollTestCase(PluginTestCase):
plugins = ('Detroll',)

def testCharsetFromContentType(self):
# Regression test for issue #2: the declared charset must be pulled
# out of the Content-Type header so the body can be decoded before
# being parsed (otherwise lxml assumes latin-1 and mangles unicode
# titles, e.g. "GitHub · Social Coding").
cb = self.irc.getCallback('Detroll')
self.assertEqual(
cb.charset_from_contenttype('text/html; charset=utf-8'), 'utf-8')
self.assertEqual(
cb.charset_from_contenttype('text/html;charset=ISO-8859-1'),
'ISO-8859-1')
self.assertEqual(
cb.charset_from_contenttype('text/html; charset="utf-8"'), 'utf-8')
self.assertEqual(cb.charset_from_contenttype('text/html'), None)
self.assertEqual(cb.charset_from_contenttype(None), None)
# A utf-8 body decoded with the declared charset yields the correct
# middle dot instead of the "·" mojibake reported in the issue.
body = u'GitHub \xb7 Social Coding'.encode('utf-8')
self.assertEqual(body.decode('utf-8'), u'GitHub \xb7 Social Coding')


# vim:set shiftwidth=4 tabstop=4 expandtab textwidth=79: