Kaydet (Commit) 689a5580 authored tarafından Benjamin Peterson's avatar Benjamin Peterson

in tokenize.detect_encoding(), return utf-8-sig when a BOM is found

üst 8c804273
...@@ -95,7 +95,8 @@ function it uses to do this is available: ...@@ -95,7 +95,8 @@ function it uses to do this is available:
It detects the encoding from the presence of a UTF-8 BOM or an encoding It detects the encoding from the presence of a UTF-8 BOM or an encoding
cookie as specified in :pep:`263`. If both a BOM and a cookie are present, cookie as specified in :pep:`263`. If both a BOM and a cookie are present,
but disagree, a SyntaxError will be raised. but disagree, a SyntaxError will be raised. Note that if the BOM is found,
``'utf-8-sig'`` will be returned as an encoding.
If no encoding is specified, then the default of ``'utf-8'`` will be returned. If no encoding is specified, then the default of ``'utf-8'`` will be returned.
......
...@@ -726,7 +726,7 @@ class TestDetectEncoding(TestCase): ...@@ -726,7 +726,7 @@ class TestDetectEncoding(TestCase):
b'do_something(else)\n' b'do_something(else)\n'
) )
encoding, consumed_lines = detect_encoding(self.get_readline(lines)) encoding, consumed_lines = detect_encoding(self.get_readline(lines))
self.assertEquals(encoding, 'utf-8') self.assertEquals(encoding, 'utf-8-sig')
self.assertEquals(consumed_lines, self.assertEquals(consumed_lines,
[b'# something\n', b'print(something)\n']) [b'# something\n', b'print(something)\n'])
...@@ -747,7 +747,7 @@ class TestDetectEncoding(TestCase): ...@@ -747,7 +747,7 @@ class TestDetectEncoding(TestCase):
b'do_something(else)\n' b'do_something(else)\n'
) )
encoding, consumed_lines = detect_encoding(self.get_readline(lines)) encoding, consumed_lines = detect_encoding(self.get_readline(lines))
self.assertEquals(encoding, 'utf-8') self.assertEquals(encoding, 'utf-8-sig')
self.assertEquals(consumed_lines, [b'# coding=utf-8\n']) self.assertEquals(consumed_lines, [b'# coding=utf-8\n'])
def test_mismatched_bom_and_cookie_first_line_raises_syntaxerror(self): def test_mismatched_bom_and_cookie_first_line_raises_syntaxerror(self):
...@@ -779,7 +779,7 @@ class TestDetectEncoding(TestCase): ...@@ -779,7 +779,7 @@ class TestDetectEncoding(TestCase):
b'do_something(else)\n' b'do_something(else)\n'
) )
encoding, consumed_lines = detect_encoding(self.get_readline(lines)) encoding, consumed_lines = detect_encoding(self.get_readline(lines))
self.assertEquals(encoding, 'utf-8') self.assertEquals(encoding, 'utf-8-sig')
self.assertEquals(consumed_lines, self.assertEquals(consumed_lines,
[b'#! something\n', b'f# coding=utf-8\n']) [b'#! something\n', b'f# coding=utf-8\n'])
...@@ -833,12 +833,12 @@ class TestDetectEncoding(TestCase): ...@@ -833,12 +833,12 @@ class TestDetectEncoding(TestCase):
readline = self.get_readline((b'\xef\xbb\xbfprint(something)\n',)) readline = self.get_readline((b'\xef\xbb\xbfprint(something)\n',))
encoding, consumed_lines = detect_encoding(readline) encoding, consumed_lines = detect_encoding(readline)
self.assertEquals(encoding, 'utf-8') self.assertEquals(encoding, 'utf-8-sig')
self.assertEquals(consumed_lines, [b'print(something)\n']) self.assertEquals(consumed_lines, [b'print(something)\n'])
readline = self.get_readline((b'\xef\xbb\xbf',)) readline = self.get_readline((b'\xef\xbb\xbf',))
encoding, consumed_lines = detect_encoding(readline) encoding, consumed_lines = detect_encoding(readline)
self.assertEquals(encoding, 'utf-8') self.assertEquals(encoding, 'utf-8-sig')
self.assertEquals(consumed_lines, []) self.assertEquals(consumed_lines, [])
readline = self.get_readline((b'# coding: bad\n',)) readline = self.get_readline((b'# coding: bad\n',))
......
...@@ -301,14 +301,16 @@ def detect_encoding(readline): ...@@ -301,14 +301,16 @@ def detect_encoding(readline):
in. in.
It detects the encoding from the presence of a utf-8 bom or an encoding It detects the encoding from the presence of a utf-8 bom or an encoding
cookie as specified in pep-0263. If both a bom and a cookie are present, cookie as specified in pep-0263. If both a bom and a cookie are present, but
but disagree, a SyntaxError will be raised. If the encoding cookie is an disagree, a SyntaxError will be raised. If the encoding cookie is an invalid
invalid charset, raise a SyntaxError. charset, raise a SyntaxError. Note that if a utf-8 bom is found,
'utf-8-sig' is returned.
If no encoding is specified, then the default of 'utf-8' will be returned. If no encoding is specified, then the default of 'utf-8' will be returned.
""" """
bom_found = False bom_found = False
encoding = None encoding = None
default = 'utf-8'
def read_or_stop(): def read_or_stop():
try: try:
return readline() return readline()
...@@ -340,8 +342,9 @@ def detect_encoding(readline): ...@@ -340,8 +342,9 @@ def detect_encoding(readline):
if first.startswith(BOM_UTF8): if first.startswith(BOM_UTF8):
bom_found = True bom_found = True
first = first[3:] first = first[3:]
default = 'utf-8-sig'
if not first: if not first:
return 'utf-8', [] return default, []
encoding = find_cookie(first) encoding = find_cookie(first)
if encoding: if encoding:
...@@ -349,13 +352,13 @@ def detect_encoding(readline): ...@@ -349,13 +352,13 @@ def detect_encoding(readline):
second = read_or_stop() second = read_or_stop()
if not second: if not second:
return 'utf-8', [first] return default, [first]
encoding = find_cookie(second) encoding = find_cookie(second)
if encoding: if encoding:
return encoding, [first, second] return encoding, [first, second]
return 'utf-8', [first, second] return default, [first, second]
def tokenize(readline): def tokenize(readline):
...@@ -394,6 +397,9 @@ def _tokenize(readline, encoding): ...@@ -394,6 +397,9 @@ def _tokenize(readline, encoding):
indents = [0] indents = [0]
if encoding is not None: if encoding is not None:
if encoding == "utf-8-sig":
# BOM will already have been stripped.
encoding = "utf-8"
yield TokenInfo(ENCODING, encoding, (0, 0), (0, 0), '') yield TokenInfo(ENCODING, encoding, (0, 0), (0, 0), '')
while True: # loop over lines in stream while True: # loop over lines in stream
try: try:
......
...@@ -283,6 +283,9 @@ C-API ...@@ -283,6 +283,9 @@ C-API
Library Library
------- -------
- ``tokenize.detect_encoding`` now returns ``'utf-8-sig'`` when a UTF-8 BOM is
detected.
- Issue #8024: Update the Unicode database to 5.2. - Issue #8024: Update the Unicode database to 5.2.
- Issue #6716/2: Backslash-replace error output in compilall. - Issue #6716/2: Backslash-replace error output in compilall.
......
Markdown is supported
0% or
You are about to add 0 people to the discussion. Proceed with caution.
Finish editing this message first!
Please register or to comment