Only skip invalid UTF8 tests when actually using SIMD
This commit is contained in:
parent
5d9a28f7a1
commit
3d0ec1c684
1 changed files with 9 additions and 7 deletions
|
|
@ -321,25 +321,27 @@ def test_expected(src, expected, which=2):
|
|||
# Bad continuation byte (restored as ASCII)
|
||||
pb(b'"\xe1\x28\xa1"', '"\ufffd(\ufffd"') # )
|
||||
|
||||
# The following all fail and need to be fixed in the SIMD parser
|
||||
# The following all fail when using SIMD and need to be fixed in the SIMD parser
|
||||
if which != 1:
|
||||
continue
|
||||
|
||||
# Overlong 2-byte sequence for U+0000 (should be `0x00`)
|
||||
# pb(b'"\xc0\x80"', '"\ufffd\ufffd"')
|
||||
pb(b'"\xc0\x80"', '"\ufffd\ufffd"')
|
||||
|
||||
# Overlong 3-byte sequence for U+0000 (violates boundary)
|
||||
# pb(b'"\xe0\x80\x80"', '"\ufffd\ufffd\ufffd"')
|
||||
pb(b'"\xe0\x80\x80"', '"\ufffd\ufffd\ufffd"')
|
||||
|
||||
# Overlong 4-byte sequence for U+0000 (violates boundary)
|
||||
# pb(b'"\xf0\x80\x80\x80"', '"\ufffd\ufffd\ufffd\ufffd"')
|
||||
pb(b'"\xf0\x80\x80\x80"', '"\ufffd\ufffd\ufffd\ufffd"')
|
||||
|
||||
# High surrogate code point
|
||||
# pb(b'"\xed\xa0\x80"', '"\ufffd\ufffd\ufffd"')
|
||||
pb(b'"\xed\xa0\x80"', '"\ufffd\ufffd\ufffd"')
|
||||
|
||||
# Low surrogate code point
|
||||
# pb(b'"\xed\xb0\x80"', '"\ufffd\ufffd\ufffd"')
|
||||
pb(b'"\xed\xb0\x80"', '"\ufffd\ufffd\ufffd"')
|
||||
|
||||
# Too large first codepoint
|
||||
# pb(b'"\xff\x80\x80\x80"', '"\ufffd\ufffd\ufffd\ufffd"')
|
||||
pb(b'"\xff\x80\x80\x80"', '"\ufffd\ufffd\ufffd\ufffd"')
|
||||
|
||||
|
||||
def test_find_either_of_two_bytes(self):
|
||||
|
|
|
|||
Loading…
Reference in a new issue