Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions NEWS
Original file line number Diff line number Diff line change
Expand Up @@ -114,6 +114,8 @@ PHP NEWS
elements. (mehmetcansahin)
. Enforce max_filter_count: limit the number of filters that can be chained
in a php://filter URL. (Sjoerd Langkemper)
. Improved performance of resolving the charset of HTML functions,
especially UTF-8. (Nicolas Grekas)

- URI:
. Fix casing of enum cases in UriHostType and UrlHostType to match the RFC
Expand Down
3 changes: 3 additions & 0 deletions UPGRADING
Original file line number Diff line number Diff line change
Expand Up @@ -100,6 +100,9 @@ PHP 8.7 UPGRADE NOTES
. Improved performance of str_rot13().
. Improved performance of pack() for large strings using the a, A and Z
formats.
. Improved performance of htmlspecialchars(), htmlentities(),
html_entity_decode() and get_html_translation_table(), especially with
UTF-8.

- MBString:
. Improved performance of mb_strlen for UTF-8 strings.
62 changes: 47 additions & 15 deletions ext/standard/html.c
Original file line number Diff line number Diff line change
Expand Up @@ -356,32 +356,64 @@ PHPAPI unsigned int php_next_utf8_char(
}
/* }}} */

/* {{{ charset_is_utf8
* Returns whether a charset name is "utf-8", in any case */
static zend_always_inline bool charset_is_utf8(const char *charset_hint)
{
return zend_tolower_ascii(charset_hint[0]) == 'u' && zend_tolower_ascii(charset_hint[1]) == 't'
&& zend_tolower_ascii(charset_hint[2]) == 'f' && charset_hint[3] == '-' && charset_hint[4] == '8'
&& charset_hint[5] == '\0';
}
/* }}} */

/* {{{ entity_charset lookup_charset
* Returns the charset identifier of a non-empty charset name, or UTF-8 if it is not supported. */
static zend_never_inline enum entity_charset lookup_charset(const char *charset_hint, bool quiet)
{
const unsigned char first = zend_tolower_ascii(charset_hint[0]);

/* now walk the charset map and look for the codeset (UTF-8 is resolved by the caller) */
for (size_t i = 0; i < sizeof(charset_map)/sizeof(charset_map[0]); i++) {
const char *codeset = charset_map[i].codeset;
Comment thread
nicolas-grekas marked this conversation as resolved.

if (zend_tolower_ascii(codeset[0]) != first) {
continue;
}
/* compare up to the NUL of codeset: a mismatch stops the loop at the NUL of charset_hint at the latest */
for (size_t j = 1; zend_tolower_ascii(charset_hint[j]) == zend_tolower_ascii(codeset[j]); j++) {
if (codeset[j] == '\0') {
return charset_map[i].charset;
}
}
}

if (!quiet) {
php_error_docref(NULL, E_WARNING, "Charset \"%s\" is not supported, assuming UTF-8",
charset_hint);
}

return cs_utf_8;
}
/* }}} */

/* {{{ entity_charset determine_charset
* Returns the charset identifier based on an explicitly provided charset,
* the internal_encoding and default_charset ini settings, or UTF-8 by default. */
static enum entity_charset determine_charset(const char *charset_hint, bool quiet)
{
if (!charset_hint || !*charset_hint) {
charset_hint = get_default_charset();
}

if (charset_hint && *charset_hint) {
size_t len = strlen(charset_hint);
/* now walk the charset map and look for the codeset */
for (size_t i = 0; i < sizeof(charset_map)/sizeof(charset_map[0]); i++) {
if (len == charset_map[i].codeset_len &&
zend_binary_strcasecmp(charset_hint, len, charset_map[i].codeset, len) == 0) {
return charset_map[i].charset;
}
if (!charset_hint) {
return cs_utf_8;
}
}

if (!quiet) {
php_error_docref(NULL, E_WARNING, "Charset \"%s\" is not supported, assuming UTF-8",
charset_hint);
}
/* UTF-8 is the most used charset, and it is not in the charset map */
if (charset_is_utf8(charset_hint)) {
return cs_utf_8;
}

return cs_utf_8;
return lookup_charset(charset_hint, quiet);
}
/* }}} */

Expand Down
1 change: 0 additions & 1 deletion ext/standard/html_tables.h
Original file line number Diff line number Diff line change
Expand Up @@ -40,7 +40,6 @@ static const struct {
{ "ISO8859-1", sizeof("ISO8859-1")-1, cs_8859_1 },
{ "ISO-8859-15", sizeof("ISO-8859-15")-1, cs_8859_15 },
{ "ISO8859-15", sizeof("ISO8859-15")-1, cs_8859_15 },
{ "utf-8", sizeof("utf-8")-1, cs_utf_8 },
{ "cp1252", sizeof("cp1252")-1, cs_cp1252 },
{ "Windows-1252", sizeof("Windows-1252")-1, cs_cp1252 },
{ "1252", sizeof("1252")-1, cs_cp1252 },
Expand Down
1 change: 0 additions & 1 deletion ext/standard/html_tables/html_table_gen.php
Original file line number Diff line number Diff line change
Expand Up @@ -60,7 +60,6 @@ enum entity_charset charset;
{ "ISO8859-1", sizeof("ISO8859-1")-1, cs_8859_1 },
{ "ISO-8859-15", sizeof("ISO-8859-15")-1, cs_8859_15 },
{ "ISO8859-15", sizeof("ISO8859-15")-1, cs_8859_15 },
{ "utf-8", sizeof("utf-8")-1, cs_utf_8 },
{ "cp1252", sizeof("cp1252")-1, cs_cp1252 },
{ "Windows-1252", sizeof("Windows-1252")-1, cs_cp1252 },
{ "1252", sizeof("1252")-1, cs_cp1252 },
Expand Down
32 changes: 32 additions & 0 deletions ext/standard/tests/strings/html_charset_utf8.phpt
Original file line number Diff line number Diff line change
@@ -0,0 +1,32 @@
--TEST--
Charset names that are UTF-8 or close to it
--FILE--
<?php
foreach (['UTF-8', 'utf-8', 'uTf-8', 'UTF8', 'UTF-', 'UTF-8 ', 'UTF-88', 'UTF_8'] as $charset) {
$escaped = htmlspecialchars("\xC3\xA9<", ENT_QUOTES, $charset);
echo var_export($charset, true), ": $escaped\n";
}

ini_set('default_charset', 'uTF-8');
echo htmlspecialchars("\xC3\xA9<", ENT_QUOTES), "\n";
?>
--EXPECTF--
'UTF-8': é&lt;
'utf-8': é&lt;
'uTf-8': é&lt;

Warning: htmlspecialchars(): Charset "UTF8" is not supported, assuming UTF-8 in %s on line %d
'UTF8': é&lt;

Warning: htmlspecialchars(): Charset "UTF-" is not supported, assuming UTF-8 in %s on line %d
'UTF-': é&lt;

Warning: htmlspecialchars(): Charset "UTF-8 " is not supported, assuming UTF-8 in %s on line %d
'UTF-8 ': é&lt;

Warning: htmlspecialchars(): Charset "UTF-88" is not supported, assuming UTF-8 in %s on line %d
'UTF-88': é&lt;

Warning: htmlspecialchars(): Charset "UTF_8" is not supported, assuming UTF-8 in %s on line %d
'UTF_8': é&lt;
é&lt;
Loading