Properly compute UTF16 sizes for escapes (#12616)

This commit is contained in:
Cardillan
2026-09-09 10:17:14 -04:00
committed by GitHub
parent 5566674c29
commit f02d919b64
+13 -6
View File
@@ -51,7 +51,7 @@ public class LParser{
//skip over \n, \" and \\ escape sequences
//this doesn't actually transform the sequences, as that would output invalid characters into Statement fields and break round-trip parsing
if(c == '\\' && pos + 1 < chars.length && (chars[pos + 1] == 'n' || chars[pos + 1] == '"' || chars[pos + 1] == '\\')){
utflen += 2;
utflen += utf16size(chars[pos + 1]);
pos ++; //consume the escaped character too
continue;
}
@@ -59,10 +59,13 @@ public class LParser{
//uXXXX: validate 4 hex digits
if(c == '\\' && pos + 1 < chars.length && chars[pos + 1] == 'u'){
if(pos + 5 >= chars.length) error("Invalid \\u escape; expected 4 hex digits.");
int value = 0;
for(int j = pos + 2; j <= pos + 5; j++){
if(Character.digit(chars[j], 16) == -1) error("Invalid \\u escape; expected 4 hex digits.");
//if any of the digits is invalid, digit() returns -1 and value becomes and stays negative
value = value << 4 | Character.digit(chars[j], 16);
}
utflen += 3; //decoded char may take up to 3 bytes in modified UTF-8
if(value < 0) error("Invalid \\u escape; expected 4 hex digits.");
utflen += utf16size(value);
pos += 5; //consume u and the 4 hex digits
continue;
}
@@ -73,8 +76,7 @@ public class LParser{
break;
}
// See ByteBufferOutput.writeUTF()
utflen += c != 0 && c <= 0x7F ? 1 : c <= 0x7FF ? 2 : 3;
utflen += utf16size(c);
}
if(pos >= chars.length || chars[pos] != '"') error("Missing closing quote \" before end of file.");
@@ -85,6 +87,11 @@ public class LParser{
return new String(chars, from, pos - from);
}
static int utf16size(int c){
//see ByteBufferOutput.writeUTF()
return c != 0 && c < 0x80 ? 1 : c < 0x800 ? 2 : 3;
}
String token(){
int from = pos;
@@ -243,4 +250,4 @@ public class LParser{
}
}
}
}