mirror of
https://github.com/lua/lua.git
synced 2026-07-23 14:39:05 +00:00
BUG: shift overflow in utf-8 decode
An initial byte \xFF will ask for 7 continuation bytes, and then the shift by (count * 5) will try to shift 35 bits.
This commit is contained in:
@@ -56,6 +56,8 @@ static const char *utf8_decode (const char *s, l_uint32 *val, int strict) {
|
||||
l_uint32 res = 0; /* final result */
|
||||
if (c < 0x80) /* ASCII? */
|
||||
res = c;
|
||||
else if (c >= 0xfe) /* c >= 1111 1110b ? */
|
||||
return NULL; /* would need six or more continuation bytes */
|
||||
else {
|
||||
int count = 0; /* to count number of continuation bytes */
|
||||
for (; c & 0x40; c <<= 1) { /* while it needs continuation bytes... */
|
||||
@@ -64,8 +66,9 @@ static const char *utf8_decode (const char *s, l_uint32 *val, int strict) {
|
||||
return NULL; /* invalid byte sequence */
|
||||
res = (res << 6) | (cc & 0x3F); /* add lower 6 bits from cont. byte */
|
||||
}
|
||||
lua_assert(count <= 5);
|
||||
res |= ((l_uint32)(c & 0x7F) << (count * 5)); /* add first byte */
|
||||
if (count > 5 || res > MAXUTF || res < limits[count])
|
||||
if (res > MAXUTF || res < limits[count])
|
||||
return NULL; /* invalid byte sequence */
|
||||
s += count; /* skip continuation bytes read */
|
||||
}
|
||||
|
||||
2
makefile
2
makefile
@@ -60,7 +60,7 @@ CWARNS= $(CWARNSCPP) $(CWARNSC) $(CWARNGCC)
|
||||
# create problems; some are only available in newer gcc versions. To
|
||||
# use some of them, we also have to define an environment variable
|
||||
# ASAN_OPTIONS="detect_invalid_pointer_pairs=2".
|
||||
# -fsanitize=undefined
|
||||
# -fsanitize=undefined (you may need to add "-lubsan" to libs)
|
||||
# -fsanitize=pointer-subtract -fsanitize=address -fsanitize=pointer-compare
|
||||
# TESTS= -DLUA_USER_H='"ltests.h"' -Og -g
|
||||
|
||||
|
||||
@@ -238,10 +238,18 @@ s = "\0 \x7F\z
|
||||
s = string.gsub(s, " ", "")
|
||||
check(s, {0,0x7F, 0x80,0x7FF, 0x800,0xFFFF, 0x10000,0x10FFFF})
|
||||
|
||||
|
||||
-- again, without strictness
|
||||
s = "\xF0\x90\x80\x80 \xF7\xBF\xBF\xBF\z
|
||||
\xF8\x88\x80\x80\x80 \xFB\xBF\xBF\xBF\xBF\z
|
||||
\xFC\x84\x80\x80\x80\x80 \xFD\xBF\xBF\xBF\xBF\xBF"
|
||||
s = string.gsub(s, " ", "")
|
||||
check(s, {0x10000,0x1FFFFF, 0x200000,0x3FFFFFF, 0x4000000,0x7FFFFFFF}, true)
|
||||
|
||||
do
|
||||
-- original UTF-8 values
|
||||
local s = "\u{4000000}\u{7FFFFFFF}"
|
||||
assert(#s == 12)
|
||||
assert(s == "\xFC\x84\x80\x80\x80\x80\xFD\xBF\xBF\xBF\xBF\xBF")
|
||||
check(s, {0x4000000, 0x7FFFFFFF}, true)
|
||||
|
||||
s = "\u{200000}\u{3FFFFFF}"
|
||||
@@ -257,6 +265,10 @@ local x = "日本語a-4\0éó"
|
||||
check(x, {26085, 26412, 35486, 97, 45, 52, 0, 233, 243})
|
||||
|
||||
|
||||
-- more than 5 continuation bytes
|
||||
assert(not utf8.len("\xff\x8f\x8f\x8f\x8f\x8f\x8f\x8f"))
|
||||
|
||||
|
||||
-- Supplementary Characters
|
||||
check("𣲷𠜎𠱓𡁻𠵼ab𠺢",
|
||||
{0x23CB7, 0x2070E, 0x20C53, 0x2107B, 0x20D7C, 0x61, 0x62, 0x20EA2,})
|
||||
|
||||
Reference in New Issue
Block a user