Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 7 additions & 0 deletions NEWS
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,13 @@ PHP NEWS
registrations are freed while still reachable from the cycle collector.
(Ilia Alshanetsky)

- Lexbor:
. Merge patches 859f100, 8a14bc0, f67ce4b, a36e09a and b0f7412, fixing a heap
buffer overflow in :lexbor-contains() parsing, buffer overflows in malformed
decode replay, uninitialized memory in IDNA buffer growth, dropped usernames
containing an at sign and the URLSearchParams tail pointer.
(alexandre-daubois)


24 Sep 2026, PHP 8.5.11

Expand Down
16 changes: 16 additions & 0 deletions ext/dom/tests/modern/css_selectors/lexbor_contains.phpt
Original file line number Diff line number Diff line change
@@ -0,0 +1,16 @@
--TEST--
CSS Selectors - Pseudo classes: :lexbor-contains() with an argument longer than its string header
--EXTENSIONS--
dom
--FILE--
<?php

$dom = Dom\HTMLDocument::createFromString('<p>needle</p>', LIBXML_NOERROR);

var_dump($dom->querySelectorAll(':lexbor-contains("' . str_repeat('needle', 1024) . '")')->length);
var_dump($dom->querySelectorAll(':lexbor-contains("needle")')->length);

?>
--EXPECT--
int(0)
int(0)
5 changes: 2 additions & 3 deletions ext/lexbor/lexbor/css/selectors/pseudo_state.c
Original file line number Diff line number Diff line change
Expand Up @@ -227,13 +227,12 @@ lxb_css_selectors_state_pseudo_class_function_lexbor_contains(lxb_css_parser_t *
contains->insensitive = false;
str = &contains->str;

str->data = lexbor_mraw_alloc(parser->memory->mraw,
sizeof(lexbor_str_t));
str->data = lexbor_mraw_alloc(parser->memory->mraw, length + 1);
if (str->data == NULL) {
return lxb_css_parser_memory_fail(parser);
}

memcpy(str->data, data, length + 1);
memcpy(str->data, data, length);

str->length = length;
str->data[length] = '\0';
Expand Down
37 changes: 33 additions & 4 deletions ext/lexbor/lexbor/encoding/decode.c
Original file line number Diff line number Diff line change
Expand Up @@ -912,6 +912,13 @@ lxb_encoding_decode_iso_2022_jp(lxb_encoding_decode_t *ctx,
}
LXB_ENCODING_DECODE_ERROR_END();

if (ctx->buffer_used >= ctx->buffer_length) {
iso->prepand = iso->lead;
iso->lead = 0x00;

return LXB_STATUS_SMALL_BUFFER;
}

byte = iso->lead;
iso->lead = 0x00;

Expand Down Expand Up @@ -1279,6 +1286,12 @@ lxb_encoding_decode_utf_16(lxb_encoding_decode_t *ctx, bool is_be,
}
LXB_ENCODING_DECODE_ERROR_END();

if (ctx->buffer_used >= ctx->buffer_length) {
ctx->u.lead = lead + 0x01;

return LXB_STATUS_SMALL_BUFFER;
}

goto lead_state;
}

Expand Down Expand Up @@ -1723,6 +1736,13 @@ lxb_encoding_decode_gb18030(lxb_encoding_decode_t *ctx,
}
LXB_ENCODING_DECODE_ERROR_END();

if (ctx->buffer_used >= ctx->buffer_length) {
ctx->prepend = true;
ctx->u.gb18030.first = second;

return LXB_STATUS_SMALL_BUFFER;
}

first = second;

goto prepend_first;
Expand Down Expand Up @@ -1756,11 +1776,8 @@ lxb_encoding_decode_gb18030(lxb_encoding_decode_t *ctx,
}
LXB_ENCODING_DECODE_ERROR_END();

LXB_ENCODING_DECODE_APPEND_WO_CHECK(ctx, second);

if (ctx->buffer_used == ctx->buffer_length) {
if (ctx->buffer_used >= ctx->buffer_length) {
ctx->prepend = true;
ctx->have_error = true;

/* First is a fake for trigger */
ctx->u.gb18030.first = 0x01;
Expand All @@ -1770,6 +1787,18 @@ lxb_encoding_decode_gb18030(lxb_encoding_decode_t *ctx,
return LXB_STATUS_SMALL_BUFFER;
}

LXB_ENCODING_DECODE_APPEND_WO_CHECK(ctx, second);

if (ctx->buffer_used >= ctx->buffer_length) {
ctx->prepend = true;

ctx->u.gb18030.first = third;
ctx->u.gb18030.second = 0x00;
ctx->u.gb18030.third = 0x00;

return LXB_STATUS_SMALL_BUFFER;
}

first = third;

goto prepend_first;
Expand Down
28 changes: 19 additions & 9 deletions ext/lexbor/lexbor/unicode/idna.c
Original file line number Diff line number Diff line change
Expand Up @@ -117,12 +117,14 @@ lxb_unicode_idna_realloc(lxb_codepoint_t *buf, const lxb_codepoint_t *buffer,
lxb_codepoint_t *tmp;

nlen = ((*buf_end - buf) * 4) + len;

if (buf == buffer) {
tmp = lexbor_malloc(nlen * sizeof(lxb_codepoint_t));
if (tmp == NULL) {
return NULL;
}

memcpy(tmp, buf, (*buf_p - buf) * sizeof(lxb_codepoint_t));
}
else {
tmp = lexbor_realloc(buf, nlen * sizeof(lxb_codepoint_t));
Expand Down Expand Up @@ -458,13 +460,17 @@ lxb_unicode_idna_ascii_puny_cb(const lxb_char_t *data, size_t length, void *ctx,

if (asc->buf == asc->buffer) {
tmp = lexbor_malloc(nlen);
if (tmp == NULL) {
return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
}

memcpy(tmp, asc->buf, asc->p - asc->buf);
}
else {
tmp = lexbor_realloc(asc->buf, nlen);
}

if (tmp == NULL) {
return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
if (tmp == NULL) {
return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
}
}

asc->p = tmp + (asc->p - asc->buf);
Expand Down Expand Up @@ -711,13 +717,17 @@ lxb_unicode_idna_to_unicode_cb(const lxb_codepoint_t *part, size_t len,

if (asc->buf == asc->buffer) {
tmp = lexbor_malloc(nlen);
if (tmp == NULL) {
return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
}

memcpy(tmp, asc->buf, asc->p - asc->buf);
}
else {
tmp = lexbor_realloc(asc->buf, nlen);
}

if (tmp == NULL) {
return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
if (tmp == NULL) {
return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
}
}

asc->p = tmp + (asc->p - asc->buf);
Expand Down
19 changes: 9 additions & 10 deletions ext/lexbor/lexbor/url/url.c
Original file line number Diff line number Diff line change
Expand Up @@ -1753,16 +1753,13 @@ lxb_url_parse_basic_h(lxb_url_parser_t *parser, lxb_url_t *url,
break;
}

if (pswd == NULL || !at_sign) {
tmp = (pswd != NULL) ? pswd - 1 : p;

if (tmp > begin) {
status = lxb_url_percent_encode_after_utf_8(begin, tmp,
&url->username, url->mraw,
LXB_URL_MAP_USERINFO, false);
if (status != LXB_STATUS_OK) {
lxb_url_parse_return(orig_data, buf, status);
}
tmp = (pswd != NULL) ? pswd - 1 : p;
if (tmp > begin) {
status = lxb_url_percent_encode_after_utf_8(begin, tmp,
&url->username, url->mraw,
LXB_URL_MAP_USERINFO, false);
if (status != LXB_STATUS_OK) {
lxb_url_parse_return(orig_data, buf, status);
}
}

Expand Down Expand Up @@ -5106,6 +5103,8 @@ lxb_url_search_params_parse(lxb_url_search_params_t *search_params,
return status;
}

last = entry;

lexbor_str_init(&entry->value, mraw, 0);
if (entry->value.data == NULL) {
return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
Expand Down
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
From: Niels Dossche <7771979+nielsdos@users.noreply.github.com>
Date: Sat, 26 Aug 2023 15:08:59 +0200
Subject: [PATCH 01/10] Expose line and column information for use in PHP
Subject: [PATCH 01/15] Expose line and column information for use in PHP

---
source/lexbor/dom/interfaces/node.h | 2 ++
Expand Down
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
From: Niels Dossche <7771979+nielsdos@users.noreply.github.com>
Date: Mon, 14 Aug 2023 20:18:51 +0200
Subject: [PATCH 02/10] Track implied added nodes for options use in PHP
Subject: [PATCH 02/15] Track implied added nodes for options use in PHP

---
source/lexbor/html/tree.h | 3 +++
Expand Down
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
From: Niels Dossche <7771979+nielsdos@users.noreply.github.com>
Date: Thu, 24 Aug 2023 22:57:48 +0200
Subject: [PATCH 03/10] Patch utilities and data structure to be able to
Subject: [PATCH 03/15] Patch utilities and data structure to be able to
generate smaller lookup tables

Changed the generation script to check if everything fits in 32-bits.
Expand Down
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
From: Niels Dossche <7771979+nielsdos@users.noreply.github.com>
Date: Wed, 29 Nov 2023 21:26:47 +0100
Subject: [PATCH 04/10] Remove unused upper case tag static data
Subject: [PATCH 04/15] Remove unused upper case tag static data

---
source/lexbor/tag/res.h | 2 ++
Expand Down
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
From: Niels Dossche <7771979+nielsdos@users.noreply.github.com>
Date: Wed, 29 Nov 2023 21:29:31 +0100
Subject: [PATCH 05/10] Shrink size of static binary search tree
Subject: [PATCH 05/15] Shrink size of static binary search tree

This also makes it more efficient on the data cache.
---
Expand Down
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
From: Niels Dossche <7771979+nielsdos@users.noreply.github.com>
Date: Sun, 7 Jan 2024 21:59:28 +0100
Subject: [PATCH 06/10] Patch out unused CSS style code
Subject: [PATCH 06/15] Patch out unused CSS style code

---
source/lexbor/css/rule.h | 2 ++
Expand Down
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
From: Alexander Borisov <lex.borisov@gmail.com>
Date: Fri, 26 Jun 2026 18:55:56 +0300
Subject: [PATCH 07/10] URL: fixed setters for empty hosts.
Subject: [PATCH 07/15] URL: fixed setters for empty hosts.
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Expand Down
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
From: Alexander Borisov <lex.borisov@gmail.com>
Date: Fri, 5 Jun 2026 22:13:32 +0300
Subject: [PATCH 08/10] URL: fixed uninitialized memory in the path buffer
Subject: [PATCH 08/15] URL: fixed uninitialized memory in the path buffer
growth.

When a path was long enough to outgrow the on-stack buffer, the first
Expand Down
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?M=C3=A1t=C3=A9=20Kocsis?= <kocsismate@woohoolabs.com>
Date: Thu, 9 Jul 2026 21:51:05 +0200
Subject: [PATCH 09/10] Fix parsing for URL containing empty host and userinfo
Subject: [PATCH 09/15] Fix parsing for URL containing empty host and userinfo

The returned error code (LXB_URL_ERROR_TYPE_INVALID_CREDENTIALS) apparently contradicts the specification:

Expand Down
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?M=C3=A1t=C3=A9=20Kocsis?= <kocsismate@woohoolabs.com>
Date: Fri, 10 Jul 2026 22:31:16 +0200
Subject: [PATCH 10/10] Percent-encode the caret in the path
Subject: [PATCH 10/15] Percent-encode the caret in the path

The caret (^) is part of the path percent-encode set:

Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,29 @@
From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
From: Alexander Borisov <lex.borisov@gmail.com>
Date: Fri, 5 Jun 2026 21:46:26 +0300
Subject: [PATCH 11/15] URL: fixed tail pointer in URLSearchParams for
delimiter-free query.

When a query had a single token without '=' or '&' (e.g. "?abc"), the
internal tail pointer wasn't updated, so a later append() could lose the
added parameter (and write through a stale pointer). Fixed by keeping the
tail pointer in sync.

Per report from Xiansheng Cao (@HMF2021)
---
source/lexbor/url/url.c | 2 ++
1 file changed, 2 insertions(+)

diff --git a/source/lexbor/url/url.c b/source/lexbor/url/url.c
index de19239..fcae2d6 100644
--- a/source/lexbor/url/url.c
+++ b/source/lexbor/url/url.c
@@ -5106,6 +5106,8 @@ lxb_url_search_params_parse(lxb_url_search_params_t *search_params,
return status;
}

+ last = entry;
+
lexbor_str_init(&entry->value, mraw, 0);
if (entry->value.data == NULL) {
return LXB_STATUS_ERROR_MEMORY_ALLOCATION;
Original file line number Diff line number Diff line change
@@ -0,0 +1,35 @@
From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
From: Alexander Borisov <lex.borisov@gmail.com>
Date: Fri, 5 Jun 2026 22:34:23 +0300
Subject: [PATCH 12/15] CSS: fixed heap buffer overflow in :lexbor-contains()
parsing.

The contains string buffer was allocated by the size of the string
structure instead of the content length, so any value longer than
that overflowed the buffer.

Per report from Xiansheng Cao (@HMF2021)
---
source/lexbor/css/selectors/pseudo_state.c | 5 ++---
1 file changed, 2 insertions(+), 3 deletions(-)

diff --git a/source/lexbor/css/selectors/pseudo_state.c b/source/lexbor/css/selectors/pseudo_state.c
index 263ca52..2321ddf 100644
--- a/source/lexbor/css/selectors/pseudo_state.c
+++ b/source/lexbor/css/selectors/pseudo_state.c
@@ -227,13 +227,12 @@ again:
contains->insensitive = false;
str = &contains->str;

- str->data = lexbor_mraw_alloc(parser->memory->mraw,
- sizeof(lexbor_str_t));
+ str->data = lexbor_mraw_alloc(parser->memory->mraw, length + 1);
if (str->data == NULL) {
return lxb_css_parser_memory_fail(parser);
}

- memcpy(str->data, data, length + 1);
+ memcpy(str->data, data, length);

str->length = length;
str->data[length] = '\0';
Loading
Loading