Skip to content
Open
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
34 changes: 32 additions & 2 deletions ext/standard/html.c
Original file line number Diff line number Diff line change
Expand Up @@ -84,7 +84,7 @@ static char *get_default_charset(void) {
/* }}} */

/* {{{ get_next_char */
static inline unsigned int get_next_char(
static zend_always_inline unsigned int get_next_char(

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Forcing this inline also pulls it into the ambiguous-entity lookahead in find_entity_for_char(), which is cold. Do you have the overall html.o text size delta?

Copy link
Copy Markdown
Contributor Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Thanks for the fuzzing and the LIMIT_ALL catch, I'll fix the description.

html.o text grows by 2.2 KB (37,445 → 39,620): the big loop goes from 3.7 to 6.7 KB, the standalone get_next_char (1.5 KB) disappears, php_next_utf8_char grows to 757 bytes.

The lookahead copy is ~700 bytes of that. I tried keeping it out of line via a never_inline wrapper, but the wrapper alone is 1.5 KB, so the object only gets bigger, and that
build was slower across most of the suite (GCC reshuffles the registers in the big function). So I'd rather keep it as is.

enum entity_charset charset,
const unsigned char *str,
size_t str_len,
Expand Down Expand Up @@ -1125,6 +1125,12 @@ static inline void find_entity_for_char_basic(
}
/* }}} */

/* ASCII other than the characters htmlspecialchars() can escape */
static zend_always_inline bool is_non_special_ascii(unsigned char c)
{
return c < 0x80 && c != '&' && c != '"' && c != '\'' && c != '<' && c != '>';
}

/* {{{ php_escape_html_entities */
PHPAPI zend_string *php_escape_html_entities_ex(const unsigned char *old, size_t oldlen, int all, int flags, const char *hint_charset, bool double_encode, bool quiet)
{
Expand Down Expand Up @@ -1178,12 +1184,36 @@ PHPAPI zend_string *php_escape_html_entities_ex(const unsigned char *old, size_t
replaced = zend_string_alloc(maxlen, 0);
len = 0;
cursor = 0;
/* htmlspecialchars() without ENT_DISALLOWED leaves such ASCII as is */
const bool copy_ascii_runs = !all && !(flags & ENT_HTML_SUBSTITUTE_DISALLOWED_CHARS);

while (cursor < oldlen) {
const unsigned char *mbsequence = NULL;
size_t mbseqlen = 0,
cursor_before = cursor;
zend_result status = SUCCESS;
unsigned int this_char = get_next_char(charset, old, oldlen, &cursor, &status);
unsigned int this_char = old[cursor];

if (this_char >= 0x80) {
this_char = get_next_char(charset, old, oldlen, &cursor, &status);
} else if (copy_ascii_runs && is_non_special_ascii(this_char)) {
/* The run can be as long as the rest of the input */
if (maxlen - len < oldlen - cursor + 40) {
replaced = zend_string_safe_realloc(replaced, maxlen, 1, oldlen - cursor + 128, 0);
maxlen += oldlen - cursor + 128;
}
const unsigned char *in = &old[cursor], *const end = &old[oldlen];
char *out = &ZSTR_VAL(replaced)[len];
do {
*out++ = *in++;
} while (in < end && is_non_special_ascii(*in));
len = out - ZSTR_VAL(replaced);
cursor = in - old;
continue;
} else {
/* ASCII is a single byte in every charset */
cursor++;
}

/* guarantee we have at least 40 bytes to write.
* In HTML5, entities may take up to 33 bytes */
Expand Down
Loading