From cd7173b75f55a057154e27f770fecb66dac912ae Mon Sep 17 00:00:00 2001 From: ANSHUL SINGH <72524975+ekanshul@users.noreply.github.com> Date: Wed, 30 Sep 2026 17:40:10 +0000 Subject: [PATCH] Stop treating ';' as part of unquoted meta charset values The HTML spec stops an unquoted charset at ASCII whitespace or a semicolon. ContentAttrParser only stopped at spaces, so values like charset=iso-8859-2;text/html were treated as an invalid encoding. Fixes #92 --- AUTHORS.rst | 1 + CHANGES.rst | 3 +++ html5lib/_inputstream.py | 5 ++++- html5lib/tests/test_encoding.py | 15 +++++++++++++++ 4 files changed, 23 insertions(+), 1 deletion(-) diff --git a/AUTHORS.rst b/AUTHORS.rst index 90401390..2d44c8a1 100644 --- a/AUTHORS.rst +++ b/AUTHORS.rst @@ -63,4 +63,5 @@ Patches and suggestions - Ville Skyttä - Hugo van Kemenade - Mark Vasilkov +- Anshul Singh diff --git a/CHANGES.rst b/CHANGES.rst index 47dcda3a..1590b49e 100644 --- a/CHANGES.rst +++ b/CHANGES.rst @@ -20,6 +20,9 @@ Bug fixes: * The sanitizer now permits ```` tags. It used to allow ``
`` already. (#423) +* Stop treating ``;`` as part of an unquoted charset in a meta + ``Content-Type``. ``charset=iso-8859-2;text/html`` is now decoded as + ``iso-8859-2``. (#92) 1.1 ~~~ diff --git a/html5lib/_inputstream.py b/html5lib/_inputstream.py index a93b5a4e..3daa414e 100644 --- a/html5lib/_inputstream.py +++ b/html5lib/_inputstream.py @@ -15,6 +15,9 @@ # Non-unicode versions of constants for use in the pre-parser spaceCharactersBytes = frozenset([item.encode("ascii") for item in spaceCharacters]) +# Unquoted charset values in a meta Content-Type stop at ASCII whitespace or ';' +# https://html.spec.whatwg.org/#algorithm-for-extracting-a-character-encoding-from-a-meta-element +contentCharsetTerminators = spaceCharactersBytes | frozenset([b";"]) asciiLettersBytes = frozenset([item.encode("ascii") for item in asciiLetters]) asciiUppercaseBytes = frozenset([item.encode("ascii") for item in asciiUppercase]) spacesAngleBrackets = spaceCharactersBytes | frozenset([b">", b"<"]) @@ -891,7 +894,7 @@ def parse(self): # Unquoted value oldPosition = self.data.position try: - self.data.skipUntil(spaceCharactersBytes) + self.data.skipUntil(contentCharsetTerminators) return self.data[oldPosition:self.data.position] except StopIteration: # Return the whole remaining value diff --git a/html5lib/tests/test_encoding.py b/html5lib/tests/test_encoding.py index 47c4814a..1fb0a044 100644 --- a/html5lib/tests/test_encoding.py +++ b/html5lib/tests/test_encoding.py @@ -8,6 +8,18 @@ from html5lib import HTMLParser, _inputstream +@pytest.mark.parametrize("content, expected", [ + (b"charset=iso8859-2;text/html", b"iso8859-2"), + (b"text/html;charset=iso-8859-2", b"iso-8859-2"), + (b"text/html;charset=iso-8859-2;", b"iso-8859-2"), + (b"charset=iso-8859-2 ", b"iso-8859-2"), + (b'charset="iso-8859-2";text/html', b"iso-8859-2"), +]) +def test_content_attr_parser_charset(content, expected): + parser = _inputstream.ContentAttrParser(_inputstream.EncodingBytes(content)) + assert parser.parse() == expected + + def test_basic_prescan_length(): data = "Caf\u00E9".encode('utf-8') pad = 1024 - len(data) + 1 @@ -37,6 +49,9 @@ def test_parser_reparse(): ("iso-8859-2", b"", {"override_encoding": "iso-8859-2", "transport_encoding": "iso-8859-3"}), ("iso-8859-2", b"", {"transport_encoding": "iso-8859-2"}), ("iso-8859-2", b"", {"same_origin_parent_encoding": "iso-8859-3"}), + # Trailing ';' must not be treated as part of an unquoted charset (#92) + ("iso-8859-2", b'', {}), + ("iso-8859-2", b'', {}), ("iso-8859-2", b"", {"same_origin_parent_encoding": "iso-8859-2", "likely_encoding": "iso-8859-3"}), ("iso-8859-2", b"", {"same_origin_parent_encoding": "utf-16", "likely_encoding": "iso-8859-2"}), ("iso-8859-2", b"", {"same_origin_parent_encoding": "utf-16be", "likely_encoding": "iso-8859-2"}),