Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions AUTHORS.rst
Original file line number Diff line number Diff line change
Expand Up @@ -63,4 +63,5 @@ Patches and suggestions
- Ville Skyttä
- Hugo van Kemenade
- Mark Vasilkov
- Anshul Singh

3 changes: 3 additions & 0 deletions CHANGES.rst
Original file line number Diff line number Diff line change
Expand Up @@ -20,6 +20,9 @@ Bug fixes:

* The sanitizer now permits ``<summary>`` tags. It used to allow ``<details>``
already. (#423)
* Stop treating ``;`` as part of an unquoted charset in a meta
``Content-Type``. ``charset=iso-8859-2;text/html`` is now decoded as
``iso-8859-2``. (#92)

1.1
~~~
Expand Down
5 changes: 4 additions & 1 deletion html5lib/_inputstream.py
Original file line number Diff line number Diff line change
Expand Up @@ -15,6 +15,9 @@

# Non-unicode versions of constants for use in the pre-parser
spaceCharactersBytes = frozenset([item.encode("ascii") for item in spaceCharacters])
# Unquoted charset values in a meta Content-Type stop at ASCII whitespace or ';'
# https://html.spec.whatwg.org/#algorithm-for-extracting-a-character-encoding-from-a-meta-element
contentCharsetTerminators = spaceCharactersBytes | frozenset([b";"])
asciiLettersBytes = frozenset([item.encode("ascii") for item in asciiLetters])
asciiUppercaseBytes = frozenset([item.encode("ascii") for item in asciiUppercase])
spacesAngleBrackets = spaceCharactersBytes | frozenset([b">", b"<"])
Expand Down Expand Up @@ -891,7 +894,7 @@ def parse(self):
# Unquoted value
oldPosition = self.data.position
try:
self.data.skipUntil(spaceCharactersBytes)
self.data.skipUntil(contentCharsetTerminators)
return self.data[oldPosition:self.data.position]
except StopIteration:
# Return the whole remaining value
Expand Down
15 changes: 15 additions & 0 deletions html5lib/tests/test_encoding.py
Original file line number Diff line number Diff line change
Expand Up @@ -8,6 +8,18 @@
from html5lib import HTMLParser, _inputstream


@pytest.mark.parametrize("content, expected", [
(b"charset=iso8859-2;text/html", b"iso8859-2"),
(b"text/html;charset=iso-8859-2", b"iso-8859-2"),
(b"text/html;charset=iso-8859-2;", b"iso-8859-2"),
(b"charset=iso-8859-2 ", b"iso-8859-2"),
(b'charset="iso-8859-2";text/html', b"iso-8859-2"),
])
def test_content_attr_parser_charset(content, expected):
parser = _inputstream.ContentAttrParser(_inputstream.EncodingBytes(content))
assert parser.parse() == expected


def test_basic_prescan_length():
data = "<title>Caf\u00E9</title><!--a--><meta charset='utf-8'>".encode('utf-8')
pad = 1024 - len(data) + 1
Expand Down Expand Up @@ -37,6 +49,9 @@ def test_parser_reparse():
("iso-8859-2", b"", {"override_encoding": "iso-8859-2", "transport_encoding": "iso-8859-3"}),
("iso-8859-2", b"<meta charset=iso-8859-3>", {"transport_encoding": "iso-8859-2"}),
("iso-8859-2", b"<meta charset=iso-8859-2>", {"same_origin_parent_encoding": "iso-8859-3"}),
# Trailing ';' must not be treated as part of an unquoted charset (#92)
("iso-8859-2", b'<meta http-equiv="content-type" content="charset=iso-8859-2;text/html">', {}),
("iso-8859-2", b'<meta http-equiv="content-type" content="text/html;charset=iso-8859-2;">', {}),
("iso-8859-2", b"", {"same_origin_parent_encoding": "iso-8859-2", "likely_encoding": "iso-8859-3"}),
("iso-8859-2", b"", {"same_origin_parent_encoding": "utf-16", "likely_encoding": "iso-8859-2"}),
("iso-8859-2", b"", {"same_origin_parent_encoding": "utf-16be", "likely_encoding": "iso-8859-2"}),
Expand Down