Skip to content

Commit b9531ec

Browse files
ext/standard: Return the input of htmlspecialchars() when nothing needs encoding
htmlspecialchars() allocated a buffer of twice the input and decoded it character by character, even when the result was the input itself. The input is now scanned first for the characters that need encoding and for invalid multi-byte sequences, with SSE2 on 16-byte blocks. When there is none, the input string is returned, as html_entity_decode() and htmlspecialchars_decode() already do since 68dc754. Otherwise, encoding starts after the part that was scanned. The @refcount 1 annotations of htmlspecialchars() and htmlentities() are removed since they can now return their argument.
1 parent e256041 commit b9531ec

6 files changed

Lines changed: 257 additions & 95 deletions

File tree

‎Zend/Optimizer/zend_func_infos.h‎

Lines changed: 0 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -463,8 +463,6 @@ static const func_info_t func_infos[] = {
463463
F1("inet_pton", MAY_BE_STRING|MAY_BE_FALSE),
464464
F1("metaphone", MAY_BE_STRING),
465465
F1("headers_list", MAY_BE_ARRAY|MAY_BE_ARRAY_KEY_LONG|MAY_BE_ARRAY_OF_STRING),
466-
F1("htmlspecialchars", MAY_BE_STRING),
467-
F1("htmlentities", MAY_BE_STRING),
468466
F1("get_html_translation_table", MAY_BE_ARRAY|MAY_BE_ARRAY_KEY_STRING|MAY_BE_ARRAY_OF_STRING),
469467
F1("bin2hex", MAY_BE_STRING),
470468
F1("hex2bin", MAY_BE_STRING|MAY_BE_FALSE),

‎ext/standard/basic_functions.stub.php‎

Lines changed: 0 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -2267,14 +2267,12 @@ function headers_list(): array {}
22672267

22682268
/* {{{ html.c */
22692269

2270-
/** @refcount 1 */
22712270
function htmlspecialchars(string $string, int $flags = ENT_QUOTES | ENT_SUBSTITUTE | ENT_HTML401, ?string $encoding = null, bool $double_encode = true): string {}
22722271

22732272
function htmlspecialchars_decode(string $string, int $flags = ENT_QUOTES | ENT_SUBSTITUTE | ENT_HTML401): string {}
22742273

22752274
function html_entity_decode(string $string, int $flags = ENT_QUOTES | ENT_SUBSTITUTE | ENT_HTML401, ?string $encoding = null): string {}
22762275

2277-
/** @refcount 1 */
22782276
function htmlentities(string $string, int $flags = ENT_QUOTES | ENT_SUBSTITUTE | ENT_HTML401, ?string $encoding = null, bool $double_encode = true): string {}
22792277

22802278
/**

‎ext/standard/basic_functions_arginfo.h‎

Lines changed: 1 addition & 1 deletion
Some generated files are not rendered by default. Learn more about customizing how changed files appear on GitHub.

‎ext/standard/basic_functions_decl.h‎

Lines changed: 4 additions & 4 deletions
Some generated files are not rendered by default. Learn more about customizing how changed files appear on GitHub.

‎ext/standard/html.c‎

Lines changed: 205 additions & 86 deletions
Original file line numberDiff line numberDiff line change
@@ -43,6 +43,8 @@
4343
#include <locale.h>
4444

4545
#include <zend_hash.h>
46+
#include "zend_bitset.h"
47+
#include "zend_simd.h"
4648
#include "html_tables.h"
4749

4850
/* Macro for disabling flag of translation of non-basic entities where this isn't supported.
@@ -83,6 +85,90 @@ static char *get_default_charset(void) {
8385
}
8486
/* }}} */
8587

88+
/* {{{ get_next_char_utf8
89+
* UTF-8 case of get_next_char(), for a cursor before the end of str and a status set to SUCCESS */
90+
static zend_always_inline unsigned int get_next_char_utf8(
91+
const unsigned char *str,
92+
size_t str_len,
93+
size_t *cursor,
94+
zend_result *status)
95+
{
96+
size_t pos = *cursor;
97+
unsigned int this_char = 0;
98+
99+
/* We'll follow strategy 2. from section 3.6.1 of UTR #36:
100+
* "In a reported illegal byte sequence, do not include any
101+
* non-initial byte that encodes a valid character or is a leading
102+
* byte for a valid sequence." */
103+
unsigned char c;
104+
c = str[pos];
105+
if (c < 0x80) {
106+
this_char = c;
107+
pos++;
108+
} else if (c < 0xc2) {
109+
MB_FAILURE(pos, 1);
110+
} else if (c < 0xe0) {
111+
if (!CHECK_LEN(pos, 2))
112+
MB_FAILURE(pos, 1);
113+
114+
if (!utf8_trail(str[pos + 1])) {
115+
MB_FAILURE(pos, utf8_lead(str[pos + 1]) ? 1 : 2);
116+
}
117+
this_char = ((c & 0x1f) << 6) | (str[pos + 1] & 0x3f);
118+
if (this_char < 0x80) { /* non-shortest form */
119+
MB_FAILURE(pos, 2);
120+
}
121+
pos += 2;
122+
} else if (c < 0xf0) {
123+
size_t avail = str_len - pos;
124+
125+
if (avail < 3 ||
126+
!utf8_trail(str[pos + 1]) || !utf8_trail(str[pos + 2])) {
127+
if (avail < 2 || utf8_lead(str[pos + 1]))
128+
MB_FAILURE(pos, 1);
129+
else if (avail < 3 || utf8_lead(str[pos + 2]))
130+
MB_FAILURE(pos, 2);
131+
else
132+
MB_FAILURE(pos, 3);
133+
}
134+
135+
this_char = ((c & 0x0f) << 12) | ((str[pos + 1] & 0x3f) << 6) | (str[pos + 2] & 0x3f);
136+
if (this_char < 0x800) { /* non-shortest form */
137+
MB_FAILURE(pos, 3);
138+
} else if (this_char >= 0xd800 && this_char <= 0xdfff) { /* surrogate */
139+
MB_FAILURE(pos, 3);
140+
}
141+
pos += 3;
142+
} else if (c < 0xf5) {
143+
size_t avail = str_len - pos;
144+
145+
if (avail < 4 ||
146+
!utf8_trail(str[pos + 1]) || !utf8_trail(str[pos + 2]) ||
147+
!utf8_trail(str[pos + 3])) {
148+
if (avail < 2 || utf8_lead(str[pos + 1]))
149+
MB_FAILURE(pos, 1);
150+
else if (avail < 3 || utf8_lead(str[pos + 2]))
151+
MB_FAILURE(pos, 2);
152+
else if (avail < 4 || utf8_lead(str[pos + 3]))
153+
MB_FAILURE(pos, 3);
154+
else
155+
MB_FAILURE(pos, 4);
156+
}
157+
158+
this_char = ((c & 0x07) << 18) | ((str[pos + 1] & 0x3f) << 12) | ((str[pos + 2] & 0x3f) << 6) | (str[pos + 3] & 0x3f);
159+
if (this_char < 0x10000 || this_char > 0x10FFFF) { /* non-shortest form or outside range */
160+
MB_FAILURE(pos, 4);
161+
}
162+
pos += 4;
163+
} else {
164+
MB_FAILURE(pos, 1);
165+
}
166+
167+
*cursor = pos;
168+
return this_char;
169+
}
170+
/* }}} */
171+
86172
/* {{{ get_next_char */
87173
static inline unsigned int get_next_char(
88174
enum entity_charset charset,
@@ -102,76 +188,7 @@ static inline unsigned int get_next_char(
102188

103189
switch (charset) {
104190
case cs_utf_8:
105-
{
106-
/* We'll follow strategy 2. from section 3.6.1 of UTR #36:
107-
* "In a reported illegal byte sequence, do not include any
108-
* non-initial byte that encodes a valid character or is a leading
109-
* byte for a valid sequence." */
110-
unsigned char c;
111-
c = str[pos];
112-
if (c < 0x80) {
113-
this_char = c;
114-
pos++;
115-
} else if (c < 0xc2) {
116-
MB_FAILURE(pos, 1);
117-
} else if (c < 0xe0) {
118-
if (!CHECK_LEN(pos, 2))
119-
MB_FAILURE(pos, 1);
120-
121-
if (!utf8_trail(str[pos + 1])) {
122-
MB_FAILURE(pos, utf8_lead(str[pos + 1]) ? 1 : 2);
123-
}
124-
this_char = ((c & 0x1f) << 6) | (str[pos + 1] & 0x3f);
125-
if (this_char < 0x80) { /* non-shortest form */
126-
MB_FAILURE(pos, 2);
127-
}
128-
pos += 2;
129-
} else if (c < 0xf0) {
130-
size_t avail = str_len - pos;
131-
132-
if (avail < 3 ||
133-
!utf8_trail(str[pos + 1]) || !utf8_trail(str[pos + 2])) {
134-
if (avail < 2 || utf8_lead(str[pos + 1]))
135-
MB_FAILURE(pos, 1);
136-
else if (avail < 3 || utf8_lead(str[pos + 2]))
137-
MB_FAILURE(pos, 2);
138-
else
139-
MB_FAILURE(pos, 3);
140-
}
141-
142-
this_char = ((c & 0x0f) << 12) | ((str[pos + 1] & 0x3f) << 6) | (str[pos + 2] & 0x3f);
143-
if (this_char < 0x800) { /* non-shortest form */
144-
MB_FAILURE(pos, 3);
145-
} else if (this_char >= 0xd800 && this_char <= 0xdfff) { /* surrogate */
146-
MB_FAILURE(pos, 3);
147-
}
148-
pos += 3;
149-
} else if (c < 0xf5) {
150-
size_t avail = str_len - pos;
151-
152-
if (avail < 4 ||
153-
!utf8_trail(str[pos + 1]) || !utf8_trail(str[pos + 2]) ||
154-
!utf8_trail(str[pos + 3])) {
155-
if (avail < 2 || utf8_lead(str[pos + 1]))
156-
MB_FAILURE(pos, 1);
157-
else if (avail < 3 || utf8_lead(str[pos + 2]))
158-
MB_FAILURE(pos, 2);
159-
else if (avail < 4 || utf8_lead(str[pos + 3]))
160-
MB_FAILURE(pos, 3);
161-
else
162-
MB_FAILURE(pos, 4);
163-
}
164-
165-
this_char = ((c & 0x07) << 18) | ((str[pos + 1] & 0x3f) << 12) | ((str[pos + 2] & 0x3f) << 6) | (str[pos + 3] & 0x3f);
166-
if (this_char < 0x10000 || this_char > 0x10FFFF) { /* non-shortest form or outside range */
167-
MB_FAILURE(pos, 4);
168-
}
169-
pos += 4;
170-
} else {
171-
MB_FAILURE(pos, 1);
172-
}
173-
}
174-
break;
191+
return get_next_char_utf8(str, str_len, cursor, status);
175192

176193
case cs_big5:
177194
/* reference http://demo.icu-project.org/icu-bin/convexp?conv=big5 */
@@ -1125,12 +1142,90 @@ static inline void find_entity_for_char_basic(
11251142
}
11261143
/* }}} */
11271144

1128-
/* {{{ php_escape_html_entities */
1129-
PHPAPI zend_string *php_escape_html_entities_ex(const unsigned char *old, size_t oldlen, int all, int flags, const char *hint_charset, bool double_encode, bool quiet)
1145+
/* Character classes used by html_verbatim_prefix_len(). The quote classes are the flags
1146+
* that make the quotes encoded. */
1147+
#define HTML_CC_SQUOTE ENT_HTML_QUOTE_SINGLE /* ' */
1148+
#define HTML_CC_DQUOTE ENT_HTML_QUOTE_DOUBLE /* " */
1149+
#define HTML_CC_BASIC 4 /* &, < and >, always encoded */
1150+
#define HTML_CC_NON_ASCII 8 /* bytes 0x80-0xFF */
1151+
1152+
#define N HTML_CC_NON_ASCII
1153+
static const unsigned char html_char_class[256] = {
1154+
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
1155+
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
1156+
0, 0, HTML_CC_DQUOTE, 0, 0, 0, HTML_CC_BASIC, HTML_CC_SQUOTE, 0, 0, 0, 0, 0, 0, 0, 0,
1157+
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, HTML_CC_BASIC, 0, HTML_CC_BASIC, 0,
1158+
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
1159+
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
1160+
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
1161+
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
1162+
N, N, N, N, N, N, N, N, N, N, N, N, N, N, N, N,
1163+
N, N, N, N, N, N, N, N, N, N, N, N, N, N, N, N,
1164+
N, N, N, N, N, N, N, N, N, N, N, N, N, N, N, N,
1165+
N, N, N, N, N, N, N, N, N, N, N, N, N, N, N, N,
1166+
N, N, N, N, N, N, N, N, N, N, N, N, N, N, N, N,
1167+
N, N, N, N, N, N, N, N, N, N, N, N, N, N, N, N,
1168+
N, N, N, N, N, N, N, N, N, N, N, N, N, N, N, N,
1169+
N, N, N, N, N, N, N, N, N, N, N, N, N, N, N, N,
1170+
};
1171+
#undef N
1172+
1173+
/* {{{ html_verbatim_prefix_len
1174+
* Returns the length of the longest prefix of old that escape_html_entities_from() copies
1175+
* unchanged when only the basic entities are encoded and disallowed characters are not
1176+
* substituted: the prefix contains no character to encode and no invalid multi-byte
1177+
* sequence, and ends on a character boundary. */
1178+
static zend_always_inline size_t html_verbatim_prefix_len(const unsigned char *old, size_t oldlen, int flags, enum entity_charset charset)
1179+
{
1180+
/* In all supported charsets, bytes below 0x80 are single-byte characters. Bytes above
1181+
* 0x7F are characters of their own in single-byte charsets. In UTF-8 they are validated
1182+
* below, and the prefix stops at the first of them in the other multi-byte charsets. */
1183+
const unsigned char stop = HTML_CC_BASIC
1184+
| (flags & (ENT_HTML_QUOTE_SINGLE | ENT_HTML_QUOTE_DOUBLE))
1185+
| (CHARSET_SINGLE_BYTE(charset) ? 0 : HTML_CC_NON_ASCII);
1186+
size_t pos = 0;
1187+
1188+
while (1) {
1189+
#ifdef XSSE2
1190+
/* Skip blocks of 16 bytes that contain neither a character which may need encoding
1191+
* nor a byte above 0x7F. The first byte that does is checked below. */
1192+
while (oldlen - pos >= sizeof(__m128i)) {
1193+
const __m128i in = _mm_loadu_si128((const __m128i *) (old + pos));
1194+
__m128i m = _mm_or_si128(_mm_cmpeq_epi8(in, _mm_set1_epi8('&')), _mm_cmpeq_epi8(in, _mm_set1_epi8('<')));
1195+
m = _mm_or_si128(m, _mm_or_si128(_mm_cmpeq_epi8(in, _mm_set1_epi8('>')), _mm_cmpeq_epi8(in, _mm_set1_epi8('"'))));
1196+
m = _mm_or_si128(m, _mm_or_si128(_mm_cmpeq_epi8(in, _mm_set1_epi8('\'')), in));
1197+
int mask = _mm_movemask_epi8(m);
1198+
if (mask) {
1199+
pos += zend_ulong_ntz(mask);
1200+
break;
1201+
}
1202+
pos += sizeof(__m128i);
1203+
}
1204+
#endif
1205+
while (pos < oldlen && !(html_char_class[old[pos]] & stop)) {
1206+
pos++;
1207+
}
1208+
if (pos == oldlen || old[pos] < 0x80 || charset != cs_utf_8) {
1209+
return pos;
1210+
}
1211+
1212+
size_t cursor = pos;
1213+
zend_result status = SUCCESS;
1214+
get_next_char_utf8(old, oldlen, &cursor, &status);
1215+
if (status == FAILURE) {
1216+
return pos;
1217+
}
1218+
pos = cursor;
1219+
}
1220+
}
1221+
/* }}} */
1222+
1223+
/* {{{ escape_html_entities_from
1224+
* Encodes old, whose first cursor bytes are known to be copied unchanged */
1225+
static zend_string *escape_html_entities_from(const unsigned char *old, size_t oldlen, size_t cursor, int all, int flags, enum entity_charset charset, bool double_encode)
11301226
{
1131-
size_t cursor, maxlen, len;
1227+
size_t maxlen, len;
11321228
zend_string *replaced;
1133-
enum entity_charset charset = determine_charset(hint_charset, quiet);
11341229
int doctype = flags & ENT_HTML_DOC_TYPE_MASK;
11351230
entity_table_opt entity_table;
11361231
const enc_to_uni *to_uni_table = NULL;
@@ -1139,14 +1234,6 @@ PHPAPI zend_string *php_escape_html_entities_ex(const unsigned char *old, size_t
11391234
const unsigned char *replacement = NULL;
11401235
size_t replacement_len = 0;
11411236

1142-
if (all) { /* replace with all named entities */
1143-
if (!quiet && CHARSET_PARTIAL_SUPPORT(charset)) {
1144-
php_error_docref(NULL, E_NOTICE, "Only basic entities "
1145-
"substitution is supported for multi-byte encodings other than UTF-8; "
1146-
"functionality is equivalent to htmlspecialchars");
1147-
}
1148-
LIMIT_ALL(all, doctype, charset);
1149-
}
11501237
entity_table = determine_entity_table(all, doctype);
11511238
if (all && !CHARSET_UNICODE_COMPAT(charset)) {
11521239
to_uni_table = enc_to_uni_index[charset];
@@ -1176,8 +1263,8 @@ PHPAPI zend_string *php_escape_html_entities_ex(const unsigned char *old, size_t
11761263
}
11771264

11781265
replaced = zend_string_alloc(maxlen, 0);
1179-
len = 0;
1180-
cursor = 0;
1266+
memcpy(ZSTR_VAL(replaced), old, cursor);
1267+
len = cursor;
11811268
while (cursor < oldlen) {
11821269
const unsigned char *mbsequence = NULL;
11831270
size_t mbseqlen = 0,
@@ -1338,6 +1425,38 @@ PHPAPI zend_string *php_escape_html_entities_ex(const unsigned char *old, size_t
13381425
}
13391426
/* }}} */
13401427

1428+
/* {{{ php_escape_html_entities */
1429+
static zend_always_inline zend_string *escape_html_entities(const unsigned char *old, size_t oldlen, zend_string *old_str, int all, int flags, const char *hint_charset, bool double_encode, bool quiet)
1430+
{
1431+
enum entity_charset charset = determine_charset(hint_charset, quiet);
1432+
size_t verbatim_len = 0;
1433+
1434+
if (all) { /* replace with all named entities */
1435+
if (!quiet && CHARSET_PARTIAL_SUPPORT(charset)) {
1436+
php_error_docref(NULL, E_NOTICE, "Only basic entities "
1437+
"substitution is supported for multi-byte encodings other than UTF-8; "
1438+
"functionality is equivalent to htmlspecialchars");
1439+
}
1440+
LIMIT_ALL(all, flags & ENT_HTML_DOC_TYPE_MASK, charset);
1441+
}
1442+
1443+
if (!all && !(flags & ENT_HTML_SUBSTITUTE_DISALLOWED_CHARS)) {
1444+
verbatim_len = html_verbatim_prefix_len(old, oldlen, flags, charset);
1445+
if (verbatim_len == oldlen) {
1446+
/* nothing to encode */
1447+
return old_str ? zend_string_copy(old_str) : zend_string_init((const char *) old, oldlen, 0);
1448+
}
1449+
}
1450+
1451+
return escape_html_entities_from(old, oldlen, verbatim_len, all, flags, charset, double_encode);
1452+
}
1453+
1454+
PHPAPI zend_string *php_escape_html_entities_ex(const unsigned char *old, size_t oldlen, int all, int flags, const char *hint_charset, bool double_encode, bool quiet)
1455+
{
1456+
return escape_html_entities(old, oldlen, NULL, all, flags, hint_charset, double_encode, quiet);
1457+
}
1458+
/* }}} */
1459+
13411460
/* {{{ php_html_entities */
13421461
static void php_html_entities(INTERNAL_FUNCTION_PARAMETERS, int all)
13431462
{
@@ -1357,8 +1476,8 @@ static void php_html_entities(INTERNAL_FUNCTION_PARAMETERS, int all)
13571476
if (ZSTR_LEN(str) == 0) {
13581477
RETURN_EMPTY_STRING();
13591478
}
1360-
replaced = php_escape_html_entities_ex(
1361-
(unsigned char*)ZSTR_VAL(str), ZSTR_LEN(str), all, (int) flags,
1479+
replaced = escape_html_entities(
1480+
(unsigned char*)ZSTR_VAL(str), ZSTR_LEN(str), str, all, (int) flags,
13621481
hint_charset ? ZSTR_VAL(hint_charset) : NULL, double_encode, /* quiet */ 0);
13631482
RETVAL_STR(replaced);
13641483
}

0 commit comments

Comments
 (0)