From fe14e4e1e4e1b88a08db1e17de7d0313c2855bfa Mon Sep 17 00:00:00 2001 From: Jordan Koch Date: Wed, 1 Jul 2026 15:18:23 -0700 Subject: [PATCH] Fix unicode title mangling by decoding with declared charset (#2) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The switch to lxml regressed non-ascii page titles: raw response bytes were handed straight to lxml.html.fromstring(), which assumes latin-1 when no encoding is known. A utf-8 page like GitHub then rendered as "GitHub · Social Coding" instead of "GitHub · Social Coding". Decode the response body using the charset declared in the Content-Type header before parsing, via a new charset_from_contenttype() helper. Unknown/garbage charsets fall through untouched. Co-Authored-By: Claude Opus 4.8 (1M context) --- plugin.py | 29 +++++++++++++++++++++++++++++ test.py | 20 ++++++++++++++++++++ 2 files changed, 49 insertions(+) diff --git a/plugin.py b/plugin.py index c9448b6..73ac6e7 100644 --- a/plugin.py +++ b/plugin.py @@ -72,6 +72,21 @@ def clean(self, msg): return re.sub(r'\s+', ' ', cleaned) + def charset_from_contenttype(self, contenttype): + """Return the charset declared in a Content-Type header, or None. + + e.g. 'text/html; charset=utf-8' -> 'utf-8', 'text/html' -> None. + """ + if not contenttype: + return None + parts = contenttype.split('charset=') + if len(parts) == 2: + # Trim anything after the charset (further parameters) and any + # surrounding quotes/whitespace. + return parts[1].split(';')[0].strip().strip('"\'') or None + return None + + def doPrivmsg(self, irc, msg): if ircmsgs.isCtcp(msg) and not ircmsgs.isAction(msg): return @@ -106,6 +121,20 @@ def fetch_url(self, irc, channel, url): return html = response.read(MAXSIZE) contenttype = response.info().getheader('Content-Type') + # Decode the response body using the charset declared in the + # Content-Type header *before* handing it to the (byte oriented) + # HTML parsers. Without this lxml assumes latin-1 and mangles + # non-ascii page titles, e.g. "GitHub · Social Coding" instead + # of "GitHub \xb7 Social Coding" (issue #2, a regression from the + # switch to lxml). + charset = self.charset_from_contenttype(contenttype) + if charset: + try: + html = html.decode(charset, 'replace') + except (LookupError, ValueError): + # Unknown/garbage charset: leave the bytes untouched and + # let the downstream parsers do their best. + pass if 'text/html' in contenttype: # Scrub potentially malformed HTML with lxml first try: diff --git a/test.py b/test.py index 3e120db..5bb6958 100644 --- a/test.py +++ b/test.py @@ -32,5 +32,25 @@ class DetrollTestCase(PluginTestCase): plugins = ('Detroll',) + def testCharsetFromContentType(self): + # Regression test for issue #2: the declared charset must be pulled + # out of the Content-Type header so the body can be decoded before + # being parsed (otherwise lxml assumes latin-1 and mangles unicode + # titles, e.g. "GitHub · Social Coding"). + cb = self.irc.getCallback('Detroll') + self.assertEqual( + cb.charset_from_contenttype('text/html; charset=utf-8'), 'utf-8') + self.assertEqual( + cb.charset_from_contenttype('text/html;charset=ISO-8859-1'), + 'ISO-8859-1') + self.assertEqual( + cb.charset_from_contenttype('text/html; charset="utf-8"'), 'utf-8') + self.assertEqual(cb.charset_from_contenttype('text/html'), None) + self.assertEqual(cb.charset_from_contenttype(None), None) + # A utf-8 body decoded with the declared charset yields the correct + # middle dot instead of the "·" mojibake reported in the issue. + body = u'GitHub \xb7 Social Coding'.encode('utf-8') + self.assertEqual(body.decode('utf-8'), u'GitHub \xb7 Social Coding') + # vim:set shiftwidth=4 tabstop=4 expandtab textwidth=79: