Skip to content

Commit cd7173b

Browse files
committed
Stop treating ';' as part of unquoted meta charset values
The HTML spec stops an unquoted charset at ASCII whitespace or a semicolon. ContentAttrParser only stopped at spaces, so values like charset=iso-8859-2;text/html were treated as an invalid encoding. Fixes #92
1 parent fd4f032 commit cd7173b

4 files changed

Lines changed: 23 additions & 1 deletion

File tree

‎AUTHORS.rst‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -63,4 +63,5 @@ Patches and suggestions
6363
- Ville Skyttä
6464
- Hugo van Kemenade
6565
- Mark Vasilkov
66+
- Anshul Singh
6667

‎CHANGES.rst‎

Lines changed: 3 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -20,6 +20,9 @@ Bug fixes:
2020

2121
* The sanitizer now permits ``<summary>`` tags. It used to allow ``<details>``
2222
already. (#423)
23+
* Stop treating ``;`` as part of an unquoted charset in a meta
24+
``Content-Type``. ``charset=iso-8859-2;text/html`` is now decoded as
25+
``iso-8859-2``. (#92)
2326

2427
1.1
2528
~~~

‎html5lib/_inputstream.py‎

Lines changed: 4 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -15,6 +15,9 @@
1515

1616
# Non-unicode versions of constants for use in the pre-parser
1717
spaceCharactersBytes = frozenset([item.encode("ascii") for item in spaceCharacters])
18+
# Unquoted charset values in a meta Content-Type stop at ASCII whitespace or ';'
19+
# https://html.spec.whatwg.org/#algorithm-for-extracting-a-character-encoding-from-a-meta-element
20+
contentCharsetTerminators = spaceCharactersBytes | frozenset([b";"])
1821
asciiLettersBytes = frozenset([item.encode("ascii") for item in asciiLetters])
1922
asciiUppercaseBytes = frozenset([item.encode("ascii") for item in asciiUppercase])
2023
spacesAngleBrackets = spaceCharactersBytes | frozenset([b">", b"<"])
@@ -891,7 +894,7 @@ def parse(self):
891894
# Unquoted value
892895
oldPosition = self.data.position
893896
try:
894-
self.data.skipUntil(spaceCharactersBytes)
897+
self.data.skipUntil(contentCharsetTerminators)
895898
return self.data[oldPosition:self.data.position]
896899
except StopIteration:
897900
# Return the whole remaining value

‎html5lib/tests/test_encoding.py‎

Lines changed: 15 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -8,6 +8,18 @@
88
from html5lib import HTMLParser, _inputstream
99

1010

11+
@pytest.mark.parametrize("content, expected", [
12+
(b"charset=iso8859-2;text/html", b"iso8859-2"),
13+
(b"text/html;charset=iso-8859-2", b"iso-8859-2"),
14+
(b"text/html;charset=iso-8859-2;", b"iso-8859-2"),
15+
(b"charset=iso-8859-2 ", b"iso-8859-2"),
16+
(b'charset="iso-8859-2";text/html', b"iso-8859-2"),
17+
])
18+
def test_content_attr_parser_charset(content, expected):
19+
parser = _inputstream.ContentAttrParser(_inputstream.EncodingBytes(content))
20+
assert parser.parse() == expected
21+
22+
1123
def test_basic_prescan_length():
1224
data = "<title>Caf\u00E9</title><!--a--><meta charset='utf-8'>".encode('utf-8')
1325
pad = 1024 - len(data) + 1
@@ -37,6 +49,9 @@ def test_parser_reparse():
3749
("iso-8859-2", b"", {"override_encoding": "iso-8859-2", "transport_encoding": "iso-8859-3"}),
3850
("iso-8859-2", b"<meta charset=iso-8859-3>", {"transport_encoding": "iso-8859-2"}),
3951
("iso-8859-2", b"<meta charset=iso-8859-2>", {"same_origin_parent_encoding": "iso-8859-3"}),
52+
# Trailing ';' must not be treated as part of an unquoted charset (#92)
53+
("iso-8859-2", b'<meta http-equiv="content-type" content="charset=iso-8859-2;text/html">', {}),
54+
("iso-8859-2", b'<meta http-equiv="content-type" content="text/html;charset=iso-8859-2;">', {}),
4055
("iso-8859-2", b"", {"same_origin_parent_encoding": "iso-8859-2", "likely_encoding": "iso-8859-3"}),
4156
("iso-8859-2", b"", {"same_origin_parent_encoding": "utf-16", "likely_encoding": "iso-8859-2"}),
4257
("iso-8859-2", b"", {"same_origin_parent_encoding": "utf-16be", "likely_encoding": "iso-8859-2"}),

0 commit comments

Comments
 (0)