4343#include <locale.h>
4444
4545#include <zend_hash.h>
46+ #include "zend_bitset.h"
47+ #include "zend_simd.h"
4648#include "html_tables.h"
4749
4850/* Macro for disabling flag of translation of non-basic entities where this isn't supported.
@@ -83,6 +85,90 @@ static char *get_default_charset(void) {
8385}
8486/* }}} */
8587
88+ /* {{{ get_next_char_utf8
89+ * UTF-8 case of get_next_char(), for a cursor before the end of str and a status set to SUCCESS */
90+ static zend_always_inline unsigned int get_next_char_utf8 (
91+ const unsigned char * str ,
92+ size_t str_len ,
93+ size_t * cursor ,
94+ zend_result * status )
95+ {
96+ size_t pos = * cursor ;
97+ unsigned int this_char = 0 ;
98+
99+ /* We'll follow strategy 2. from section 3.6.1 of UTR #36:
100+ * "In a reported illegal byte sequence, do not include any
101+ * non-initial byte that encodes a valid character or is a leading
102+ * byte for a valid sequence." */
103+ unsigned char c ;
104+ c = str [pos ];
105+ if (c < 0x80 ) {
106+ this_char = c ;
107+ pos ++ ;
108+ } else if (c < 0xc2 ) {
109+ MB_FAILURE (pos , 1 );
110+ } else if (c < 0xe0 ) {
111+ if (!CHECK_LEN (pos , 2 ))
112+ MB_FAILURE (pos , 1 );
113+
114+ if (!utf8_trail (str [pos + 1 ])) {
115+ MB_FAILURE (pos , utf8_lead (str [pos + 1 ]) ? 1 : 2 );
116+ }
117+ this_char = ((c & 0x1f ) << 6 ) | (str [pos + 1 ] & 0x3f );
118+ if (this_char < 0x80 ) { /* non-shortest form */
119+ MB_FAILURE (pos , 2 );
120+ }
121+ pos += 2 ;
122+ } else if (c < 0xf0 ) {
123+ size_t avail = str_len - pos ;
124+
125+ if (avail < 3 ||
126+ !utf8_trail (str [pos + 1 ]) || !utf8_trail (str [pos + 2 ])) {
127+ if (avail < 2 || utf8_lead (str [pos + 1 ]))
128+ MB_FAILURE (pos , 1 );
129+ else if (avail < 3 || utf8_lead (str [pos + 2 ]))
130+ MB_FAILURE (pos , 2 );
131+ else
132+ MB_FAILURE (pos , 3 );
133+ }
134+
135+ this_char = ((c & 0x0f ) << 12 ) | ((str [pos + 1 ] & 0x3f ) << 6 ) | (str [pos + 2 ] & 0x3f );
136+ if (this_char < 0x800 ) { /* non-shortest form */
137+ MB_FAILURE (pos , 3 );
138+ } else if (this_char >= 0xd800 && this_char <= 0xdfff ) { /* surrogate */
139+ MB_FAILURE (pos , 3 );
140+ }
141+ pos += 3 ;
142+ } else if (c < 0xf5 ) {
143+ size_t avail = str_len - pos ;
144+
145+ if (avail < 4 ||
146+ !utf8_trail (str [pos + 1 ]) || !utf8_trail (str [pos + 2 ]) ||
147+ !utf8_trail (str [pos + 3 ])) {
148+ if (avail < 2 || utf8_lead (str [pos + 1 ]))
149+ MB_FAILURE (pos , 1 );
150+ else if (avail < 3 || utf8_lead (str [pos + 2 ]))
151+ MB_FAILURE (pos , 2 );
152+ else if (avail < 4 || utf8_lead (str [pos + 3 ]))
153+ MB_FAILURE (pos , 3 );
154+ else
155+ MB_FAILURE (pos , 4 );
156+ }
157+
158+ this_char = ((c & 0x07 ) << 18 ) | ((str [pos + 1 ] & 0x3f ) << 12 ) | ((str [pos + 2 ] & 0x3f ) << 6 ) | (str [pos + 3 ] & 0x3f );
159+ if (this_char < 0x10000 || this_char > 0x10FFFF ) { /* non-shortest form or outside range */
160+ MB_FAILURE (pos , 4 );
161+ }
162+ pos += 4 ;
163+ } else {
164+ MB_FAILURE (pos , 1 );
165+ }
166+
167+ * cursor = pos ;
168+ return this_char ;
169+ }
170+ /* }}} */
171+
86172/* {{{ get_next_char */
87173static inline unsigned int get_next_char (
88174 enum entity_charset charset ,
@@ -102,76 +188,7 @@ static inline unsigned int get_next_char(
102188
103189 switch (charset ) {
104190 case cs_utf_8 :
105- {
106- /* We'll follow strategy 2. from section 3.6.1 of UTR #36:
107- * "In a reported illegal byte sequence, do not include any
108- * non-initial byte that encodes a valid character or is a leading
109- * byte for a valid sequence." */
110- unsigned char c ;
111- c = str [pos ];
112- if (c < 0x80 ) {
113- this_char = c ;
114- pos ++ ;
115- } else if (c < 0xc2 ) {
116- MB_FAILURE (pos , 1 );
117- } else if (c < 0xe0 ) {
118- if (!CHECK_LEN (pos , 2 ))
119- MB_FAILURE (pos , 1 );
120-
121- if (!utf8_trail (str [pos + 1 ])) {
122- MB_FAILURE (pos , utf8_lead (str [pos + 1 ]) ? 1 : 2 );
123- }
124- this_char = ((c & 0x1f ) << 6 ) | (str [pos + 1 ] & 0x3f );
125- if (this_char < 0x80 ) { /* non-shortest form */
126- MB_FAILURE (pos , 2 );
127- }
128- pos += 2 ;
129- } else if (c < 0xf0 ) {
130- size_t avail = str_len - pos ;
131-
132- if (avail < 3 ||
133- !utf8_trail (str [pos + 1 ]) || !utf8_trail (str [pos + 2 ])) {
134- if (avail < 2 || utf8_lead (str [pos + 1 ]))
135- MB_FAILURE (pos , 1 );
136- else if (avail < 3 || utf8_lead (str [pos + 2 ]))
137- MB_FAILURE (pos , 2 );
138- else
139- MB_FAILURE (pos , 3 );
140- }
141-
142- this_char = ((c & 0x0f ) << 12 ) | ((str [pos + 1 ] & 0x3f ) << 6 ) | (str [pos + 2 ] & 0x3f );
143- if (this_char < 0x800 ) { /* non-shortest form */
144- MB_FAILURE (pos , 3 );
145- } else if (this_char >= 0xd800 && this_char <= 0xdfff ) { /* surrogate */
146- MB_FAILURE (pos , 3 );
147- }
148- pos += 3 ;
149- } else if (c < 0xf5 ) {
150- size_t avail = str_len - pos ;
151-
152- if (avail < 4 ||
153- !utf8_trail (str [pos + 1 ]) || !utf8_trail (str [pos + 2 ]) ||
154- !utf8_trail (str [pos + 3 ])) {
155- if (avail < 2 || utf8_lead (str [pos + 1 ]))
156- MB_FAILURE (pos , 1 );
157- else if (avail < 3 || utf8_lead (str [pos + 2 ]))
158- MB_FAILURE (pos , 2 );
159- else if (avail < 4 || utf8_lead (str [pos + 3 ]))
160- MB_FAILURE (pos , 3 );
161- else
162- MB_FAILURE (pos , 4 );
163- }
164-
165- this_char = ((c & 0x07 ) << 18 ) | ((str [pos + 1 ] & 0x3f ) << 12 ) | ((str [pos + 2 ] & 0x3f ) << 6 ) | (str [pos + 3 ] & 0x3f );
166- if (this_char < 0x10000 || this_char > 0x10FFFF ) { /* non-shortest form or outside range */
167- MB_FAILURE (pos , 4 );
168- }
169- pos += 4 ;
170- } else {
171- MB_FAILURE (pos , 1 );
172- }
173- }
174- break ;
191+ return get_next_char_utf8 (str , str_len , cursor , status );
175192
176193 case cs_big5 :
177194 /* reference http://demo.icu-project.org/icu-bin/convexp?conv=big5 */
@@ -1125,12 +1142,90 @@ static inline void find_entity_for_char_basic(
11251142}
11261143/* }}} */
11271144
1128- /* {{{ php_escape_html_entities */
1129- PHPAPI zend_string * php_escape_html_entities_ex (const unsigned char * old , size_t oldlen , int all , int flags , const char * hint_charset , bool double_encode , bool quiet )
1145+ /* Character classes used by html_verbatim_prefix_len(). The quote classes are the flags
1146+ * that make the quotes encoded. */
1147+ #define HTML_CC_SQUOTE ENT_HTML_QUOTE_SINGLE /* ' */
1148+ #define HTML_CC_DQUOTE ENT_HTML_QUOTE_DOUBLE /* " */
1149+ #define HTML_CC_BASIC 4 /* &, < and >, always encoded */
1150+ #define HTML_CC_NON_ASCII 8 /* bytes 0x80-0xFF */
1151+
1152+ #define N HTML_CC_NON_ASCII
1153+ static const unsigned char html_char_class [256 ] = {
1154+ 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 ,
1155+ 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 ,
1156+ 0 , 0 , HTML_CC_DQUOTE , 0 , 0 , 0 , HTML_CC_BASIC , HTML_CC_SQUOTE , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 ,
1157+ 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , HTML_CC_BASIC , 0 , HTML_CC_BASIC , 0 ,
1158+ 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 ,
1159+ 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 ,
1160+ 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 ,
1161+ 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 , 0 ,
1162+ N , N , N , N , N , N , N , N , N , N , N , N , N , N , N , N ,
1163+ N , N , N , N , N , N , N , N , N , N , N , N , N , N , N , N ,
1164+ N , N , N , N , N , N , N , N , N , N , N , N , N , N , N , N ,
1165+ N , N , N , N , N , N , N , N , N , N , N , N , N , N , N , N ,
1166+ N , N , N , N , N , N , N , N , N , N , N , N , N , N , N , N ,
1167+ N , N , N , N , N , N , N , N , N , N , N , N , N , N , N , N ,
1168+ N , N , N , N , N , N , N , N , N , N , N , N , N , N , N , N ,
1169+ N , N , N , N , N , N , N , N , N , N , N , N , N , N , N , N ,
1170+ };
1171+ #undef N
1172+
1173+ /* {{{ html_verbatim_prefix_len
1174+ * Returns the length of the longest prefix of old that escape_html_entities_from() copies
1175+ * unchanged when only the basic entities are encoded and disallowed characters are not
1176+ * substituted: the prefix contains no character to encode and no invalid multi-byte
1177+ * sequence, and ends on a character boundary. */
1178+ static zend_always_inline size_t html_verbatim_prefix_len (const unsigned char * old , size_t oldlen , int flags , enum entity_charset charset )
1179+ {
1180+ /* In all supported charsets, bytes below 0x80 are single-byte characters. Bytes above
1181+ * 0x7F are characters of their own in single-byte charsets. In UTF-8 they are validated
1182+ * below, and the prefix stops at the first of them in the other multi-byte charsets. */
1183+ const unsigned char stop = HTML_CC_BASIC
1184+ | (flags & (ENT_HTML_QUOTE_SINGLE | ENT_HTML_QUOTE_DOUBLE ))
1185+ | (CHARSET_SINGLE_BYTE (charset ) ? 0 : HTML_CC_NON_ASCII );
1186+ size_t pos = 0 ;
1187+
1188+ while (1 ) {
1189+ #ifdef XSSE2
1190+ /* Skip blocks of 16 bytes that contain neither a character which may need encoding
1191+ * nor a byte above 0x7F. The first byte that does is checked below. */
1192+ while (oldlen - pos >= sizeof (__m128i )) {
1193+ const __m128i in = _mm_loadu_si128 ((const __m128i * ) (old + pos ));
1194+ __m128i m = _mm_or_si128 (_mm_cmpeq_epi8 (in , _mm_set1_epi8 ('&' )), _mm_cmpeq_epi8 (in , _mm_set1_epi8 ('<' )));
1195+ m = _mm_or_si128 (m , _mm_or_si128 (_mm_cmpeq_epi8 (in , _mm_set1_epi8 ('>' )), _mm_cmpeq_epi8 (in , _mm_set1_epi8 ('"' ))));
1196+ m = _mm_or_si128 (m , _mm_or_si128 (_mm_cmpeq_epi8 (in , _mm_set1_epi8 ('\'' )), in ));
1197+ int mask = _mm_movemask_epi8 (m );
1198+ if (mask ) {
1199+ pos += zend_ulong_ntz (mask );
1200+ break ;
1201+ }
1202+ pos += sizeof (__m128i );
1203+ }
1204+ #endif
1205+ while (pos < oldlen && !(html_char_class [old [pos ]] & stop )) {
1206+ pos ++ ;
1207+ }
1208+ if (pos == oldlen || old [pos ] < 0x80 || charset != cs_utf_8 ) {
1209+ return pos ;
1210+ }
1211+
1212+ size_t cursor = pos ;
1213+ zend_result status = SUCCESS ;
1214+ get_next_char_utf8 (old , oldlen , & cursor , & status );
1215+ if (status == FAILURE ) {
1216+ return pos ;
1217+ }
1218+ pos = cursor ;
1219+ }
1220+ }
1221+ /* }}} */
1222+
1223+ /* {{{ escape_html_entities_from
1224+ * Encodes old, whose first cursor bytes are known to be copied unchanged */
1225+ static zend_string * escape_html_entities_from (const unsigned char * old , size_t oldlen , size_t cursor , int all , int flags , enum entity_charset charset , bool double_encode )
11301226{
1131- size_t cursor , maxlen , len ;
1227+ size_t maxlen , len ;
11321228 zend_string * replaced ;
1133- enum entity_charset charset = determine_charset (hint_charset , quiet );
11341229 int doctype = flags & ENT_HTML_DOC_TYPE_MASK ;
11351230 entity_table_opt entity_table ;
11361231 const enc_to_uni * to_uni_table = NULL ;
@@ -1139,14 +1234,6 @@ PHPAPI zend_string *php_escape_html_entities_ex(const unsigned char *old, size_t
11391234 const unsigned char * replacement = NULL ;
11401235 size_t replacement_len = 0 ;
11411236
1142- if (all ) { /* replace with all named entities */
1143- if (!quiet && CHARSET_PARTIAL_SUPPORT (charset )) {
1144- php_error_docref (NULL , E_NOTICE , "Only basic entities "
1145- "substitution is supported for multi-byte encodings other than UTF-8; "
1146- "functionality is equivalent to htmlspecialchars" );
1147- }
1148- LIMIT_ALL (all , doctype , charset );
1149- }
11501237 entity_table = determine_entity_table (all , doctype );
11511238 if (all && !CHARSET_UNICODE_COMPAT (charset )) {
11521239 to_uni_table = enc_to_uni_index [charset ];
@@ -1176,8 +1263,8 @@ PHPAPI zend_string *php_escape_html_entities_ex(const unsigned char *old, size_t
11761263 }
11771264
11781265 replaced = zend_string_alloc (maxlen , 0 );
1179- len = 0 ;
1180- cursor = 0 ;
1266+ memcpy ( ZSTR_VAL ( replaced ), old , cursor ) ;
1267+ len = cursor ;
11811268 while (cursor < oldlen ) {
11821269 const unsigned char * mbsequence = NULL ;
11831270 size_t mbseqlen = 0 ,
@@ -1338,6 +1425,38 @@ PHPAPI zend_string *php_escape_html_entities_ex(const unsigned char *old, size_t
13381425}
13391426/* }}} */
13401427
1428+ /* {{{ php_escape_html_entities */
1429+ static zend_always_inline zend_string * escape_html_entities (const unsigned char * old , size_t oldlen , zend_string * old_str , int all , int flags , const char * hint_charset , bool double_encode , bool quiet )
1430+ {
1431+ enum entity_charset charset = determine_charset (hint_charset , quiet );
1432+ size_t verbatim_len = 0 ;
1433+
1434+ if (all ) { /* replace with all named entities */
1435+ if (!quiet && CHARSET_PARTIAL_SUPPORT (charset )) {
1436+ php_error_docref (NULL , E_NOTICE , "Only basic entities "
1437+ "substitution is supported for multi-byte encodings other than UTF-8; "
1438+ "functionality is equivalent to htmlspecialchars" );
1439+ }
1440+ LIMIT_ALL (all , flags & ENT_HTML_DOC_TYPE_MASK , charset );
1441+ }
1442+
1443+ if (!all && !(flags & ENT_HTML_SUBSTITUTE_DISALLOWED_CHARS )) {
1444+ verbatim_len = html_verbatim_prefix_len (old , oldlen , flags , charset );
1445+ if (verbatim_len == oldlen ) {
1446+ /* nothing to encode */
1447+ return old_str ? zend_string_copy (old_str ) : zend_string_init ((const char * ) old , oldlen , 0 );
1448+ }
1449+ }
1450+
1451+ return escape_html_entities_from (old , oldlen , verbatim_len , all , flags , charset , double_encode );
1452+ }
1453+
1454+ PHPAPI zend_string * php_escape_html_entities_ex (const unsigned char * old , size_t oldlen , int all , int flags , const char * hint_charset , bool double_encode , bool quiet )
1455+ {
1456+ return escape_html_entities (old , oldlen , NULL , all , flags , hint_charset , double_encode , quiet );
1457+ }
1458+ /* }}} */
1459+
13411460/* {{{ php_html_entities */
13421461static void php_html_entities (INTERNAL_FUNCTION_PARAMETERS , int all )
13431462{
@@ -1357,8 +1476,8 @@ static void php_html_entities(INTERNAL_FUNCTION_PARAMETERS, int all)
13571476 if (ZSTR_LEN (str ) == 0 ) {
13581477 RETURN_EMPTY_STRING ();
13591478 }
1360- replaced = php_escape_html_entities_ex (
1361- (unsigned char * )ZSTR_VAL (str ), ZSTR_LEN (str ), all , (int ) flags ,
1479+ replaced = escape_html_entities (
1480+ (unsigned char * )ZSTR_VAL (str ), ZSTR_LEN (str ), str , all , (int ) flags ,
13621481 hint_charset ? ZSTR_VAL (hint_charset ) : NULL , double_encode , /* quiet */ 0 );
13631482 RETVAL_STR (replaced );
13641483}
0 commit comments