On Thu, Sep 24, 2026 at 7:20 AM Heikki Linnakangas <[email protected]> wrote:
>
> See also similar thread on REPEAT():
> https://www.postgresql.org/message-id/tencent_C5BBECF985A270FBC49463EDAF722CD5E005%40qq.com.

Thanks for linking to that. A lot overlap in ideas and results too.

> Whatever we do here, let's use the same implementation for REPEAT(),
> LPAD(), and RPAD().

Attached v5 moves the doubling copy into its own helper so repeat()
can use it too.

0001 is the refactor for rpad/lpad to use append_padding().

0002 adds repeat_bytes(dst, src, srclen, count) and rebuilds
append_padding() on it.

0003 replaces repeat()'s per-copy loop with a call to repeat_bytes().

repeat_bytes() copies the string once, doubles the copied region
until it is at least 16 kB, and then copies that region repeatedly
with a CHECK_FOR_INTERRUPTS() per copy.  Pure doubling in v4 was 6-8%
slower than master's loop for repeat() with a source of 100 bytes or
more and a 100 MB result, because the loop's source stays in L1 while
the doubling reads back what it just wrote.

The fixed block keeps the source in cache and puts those cases back
at parity, while still needing only a few calls for short strings.
It also keeps large results cancellable. With and without the
interrupts check was not measurable.

Timings on an AMD Ryzen 7 5700G, release build (-O3, no asserts),
pgbench -c 1, alternating before/after rounds, median ms of 5 rounds.
Pad is 'abcdefghij' repeated and cut to len. N is the result size in
bytes.  Each cell is master / patched.

# repeat(pad, N / len)
len \ N       10k            100k           1M             10M          100M
1        0.088 / 0.056  0.422 / 0.102  3.88 / 0.83    39.5 / 9.1     393 / 97
10       0.061 / 0.056  0.128 / 0.102  1.10 / 0.83    11.5 / 8.9     121 / 97
100      0.058 / 0.058  0.105 / 0.102  0.85 / 0.83    9.2 / 8.9      96 / 96
1000     0.058 / 0.059  0.106 / 0.105  0.84 / 0.83    9.2 / 9.0      95 / 95
10000    0.078 / 0.083  0.123 / 0.124  0.85 / 0.86    8.9 / 9.2      97 / 95

# rpad('x', N, pad)
len \ N       10k            100k           1M             10M          100M
1        0.106 / 0.057  0.548 / 0.061  4.80 / 0.11    53.7 / 5.7     560 / 97
10       0.105 / 0.057  0.543 / 0.061  4.83 / 0.12    52.5 / 5.6     564 / 97
100      0.107 / 0.058  0.560 / 0.062  4.89 / 0.12    53.6 / 5.5     564 / 96
1000     0.109 / 0.064  0.563 / 0.069  4.89 / 0.13    53.5 / 5.8     560 / 96
10000    0.129 / 0.107  0.576 / 0.129  5.02 / 0.18    53.2 / 5.7     566 / 97

Tests are the lpad/rpad set from v3 plus some additional ones to go
past the 16K doubling limit.

On this machine, block sizes from 16 kB up to 4 MB performed the same
within noise, while 16 MB was consistently slower for large results,
much like v4's pure doubling.  So I don't see a reason to go bigger
than 16 kB.

The doubling approach for repeat() was proposed by Chenhui Mo in [1].
Jeevan Chalke, Heikki Linnakangas, David Rowley, and Jan Nidzwetzki
also discussed and benchmarked capped/fixed-size variants there, and
that work shaped the fixed-size block used here.

This series overlaps with the repeat() work linked above, so the two are not
independent. The v2 series there, currently marked Ready for Committer,
contains the empty-result and single-byte memset() changes but leaves
the doubling change out.


Regards,
-- Sehrope Sarkuni
Founder & CEO | JackDB, Inc. | https://www.jackdb.com/
From 0c152b84af6ebed24bc65fc62682bed80db2c9a7 Mon Sep 17 00:00:00 2001
From: Nathan Bossart <[email protected]>
Date: Wed, 23 Sep 2026 11:47:02 -0500
Subject: [PATCH v5 1/3] Factor the padding loop out of lpad() and rpad().

lpad() and rpad() each carry an identical copy of the loop that
writes the padding characters.  This commit moves it into a helper
function that both call.  No functional change.

This is preparatory work for a follow-up commit that will teach the
helper to write long padding with a few large memcpy() calls instead
of one per character.

Discussion: CAH7T-apj+pFg9bRkXVGfGejS2Uu4WzdMd7KoLKTW11mdsrt1Ew@mail.gmail.com">https://postgr.es/m/CAH7T-apj+pFg9bRkXVGfGejS2Uu4WzdMd7KoLKTW11mdsrt1Ew@mail.gmail.com
---
 src/backend/utils/adt/oracle_compat.c | 60 ++++++++++++---------------
 1 file changed, 27 insertions(+), 33 deletions(-)

diff --git a/src/backend/utils/adt/oracle_compat.c b/src/backend/utils/adt/oracle_compat.c
index 7422a454397..e5238e44813 100644
--- a/src/backend/utils/adt/oracle_compat.c
+++ b/src/backend/utils/adt/oracle_compat.c
@@ -143,6 +143,31 @@ casefold(PG_FUNCTION_ARGS)
 }
 
 
+/*
+ * Append m characters of the padding string pad (padlen bytes) to dst,
+ * cycling through pad as needed, and return a pointer past the last byte
+ * written.
+ */
+static char *
+append_padding(char *dst, const char *pad, int padlen, int m)
+{
+	const char *p = pad;
+	const char *pend = pad + padlen;
+
+	while (m--)
+	{
+		int			mlen = pg_mblen_range(p, pend);
+
+		memcpy(dst, p, mlen);
+		dst += mlen;
+		p += mlen;
+		if (p == pend)			/* wrap around at end of pad */
+			p = pad;
+	}
+
+	return dst;
+}
+
 /********************************************************************
  *
  * lpad
@@ -167,10 +192,7 @@ lpad(PG_FUNCTION_ARGS)
 	text	   *string2 = PG_GETARG_TEXT_PP(2);
 	text	   *ret;
 	char	   *ptr1,
-			   *ptr2,
-			   *ptr2start,
 			   *ptr_ret;
-	const char *ptr2end;
 	int			m,
 				s1len,
 				s2len;
@@ -209,20 +231,7 @@ lpad(PG_FUNCTION_ARGS)
 
 	m = len - s1len;
 
-	ptr2 = ptr2start = VARDATA_ANY(string2);
-	ptr2end = ptr2 + s2len;
-	ptr_ret = VARDATA(ret);
-
-	while (m--)
-	{
-		int			mlen = pg_mblen_range(ptr2, ptr2end);
-
-		memcpy(ptr_ret, ptr2, mlen);
-		ptr_ret += mlen;
-		ptr2 += mlen;
-		if (ptr2 == ptr2end)	/* wrap around at end of s2 */
-			ptr2 = ptr2start;
-	}
+	ptr_ret = append_padding(VARDATA(ret), VARDATA_ANY(string2), s2len, m);
 
 	ptr1 = VARDATA_ANY(string1);
 
@@ -265,10 +274,7 @@ rpad(PG_FUNCTION_ARGS)
 	text	   *string2 = PG_GETARG_TEXT_PP(2);
 	text	   *ret;
 	char	   *ptr1,
-			   *ptr2,
-			   *ptr2start,
 			   *ptr_ret;
-	const char *ptr2end;
 	int			m,
 				s1len,
 				s2len;
@@ -320,19 +326,7 @@ rpad(PG_FUNCTION_ARGS)
 		ptr1 += mlen;
 	}
 
-	ptr2 = ptr2start = VARDATA_ANY(string2);
-	ptr2end = ptr2 + s2len;
-
-	while (m--)
-	{
-		int			mlen = pg_mblen_range(ptr2, ptr2end);
-
-		memcpy(ptr_ret, ptr2, mlen);
-		ptr_ret += mlen;
-		ptr2 += mlen;
-		if (ptr2 == ptr2end)	/* wrap around at end of s2 */
-			ptr2 = ptr2start;
-	}
+	ptr_ret = append_padding(ptr_ret, VARDATA_ANY(string2), s2len, m);
 
 	SET_VARSIZE(ret, ptr_ret - (char *) ret);
 
-- 
2.55.0

From 666fdb5b536342bcdf1edcc37e69eed39e7b13d1 Mon Sep 17 00:00:00 2001
From: Sehrope Sarkuni <[email protected]>
Date: Thu, 24 Sep 2026 11:56:07 +0000
Subject: [PATCH v5 3/3] Use repeat_bytes() in repeat()

repeat() copied the string once per repetition.  Use repeat_bytes()
instead, which writes a large result with a few doubling copies and
then block copies, and checks for interrupts between the latter as
the old loop did between copies.

Discussion: CAH7T-apj+pFg9bRkXVGfGejS2Uu4WzdMd7KoLKTW11mdsrt1Ew@mail.gmail.com">https://postgr.es/m/CAH7T-apj+pFg9bRkXVGfGejS2Uu4WzdMd7KoLKTW11mdsrt1Ew@mail.gmail.com
Discussion: https://postgr.es/m/[email protected]
---
 src/backend/utils/adt/oracle_compat.c | 12 +-----------
 src/test/regress/expected/strings.out | 19 +++++++++++++++++++
 src/test/regress/sql/strings.sql      |  5 +++++
 3 files changed, 25 insertions(+), 11 deletions(-)

diff --git a/src/backend/utils/adt/oracle_compat.c b/src/backend/utils/adt/oracle_compat.c
index 6fdbd595a9f..fb584aa8b06 100644
--- a/src/backend/utils/adt/oracle_compat.c
+++ b/src/backend/utils/adt/oracle_compat.c
@@ -1236,9 +1236,6 @@ repeat(PG_FUNCTION_ARGS)
 	text	   *result;
 	int			slen,
 				tlen;
-	int			i;
-	char	   *cp,
-			   *sp;
 
 	if (count < 0)
 		count = 0;
@@ -1255,14 +1252,7 @@ repeat(PG_FUNCTION_ARGS)
 	result = (text *) palloc(tlen);
 
 	SET_VARSIZE(result, tlen);
-	cp = VARDATA(result);
-	sp = VARDATA_ANY(string);
-	for (i = 0; i < count; i++)
-	{
-		memcpy(cp, sp, slen);
-		cp += slen;
-		CHECK_FOR_INTERRUPTS();
-	}
+	repeat_bytes(VARDATA(result), VARDATA_ANY(string), slen, count);
 
 	PG_RETURN_TEXT_P(result);
 }
diff --git a/src/test/regress/expected/strings.out b/src/test/regress/expected/strings.out
index b588398d7fe..80669ab3406 100644
--- a/src/test/regress/expected/strings.out
+++ b/src/test/regress/expected/strings.out
@@ -3474,6 +3474,25 @@ SELECT rpad('', 50000, 'abc') = (SELECT string_agg('abc', '') FROM generate_seri
  t
 (1 row)
 
+-- repeat: one, several and five copies, and empty results
+SELECT repeat('abc', 1), repeat('abc', 2), repeat('ab', 5);
+ repeat | repeat |   repeat   
+--------+--------+------------
+ abc    | abcabc | ababababab
+(1 row)
+
+SELECT repeat('abc', 0), repeat('', 5), repeat('abc', -1);
+ repeat | repeat | repeat 
+--------+--------+--------
+        |        | 
+(1 row)
+
+SELECT repeat('abc', 10000) = (SELECT string_agg('abc', '') FROM generate_series(1, 10000));
+ ?column? 
+----------
+ t
+(1 row)
+
 SELECT ltrim('zzzytrim', 'xyz');
  ltrim 
 -------
diff --git a/src/test/regress/sql/strings.sql b/src/test/regress/sql/strings.sql
index 5db0df1c6ae..1ee524543a6 100644
--- a/src/test/regress/sql/strings.sql
+++ b/src/test/regress/sql/strings.sql
@@ -1174,6 +1174,11 @@ SELECT lpad('hi', 3, 'abc'), rpad('hi', 3, 'abc');
 -- padding longer than the 16 kB block that repeat_bytes() copies
 SELECT rpad('', 50000, 'abc') = (SELECT string_agg('abc', '') FROM generate_series(1, 16666)) || 'ab';
 
+-- repeat: one, several and five copies, and empty results
+SELECT repeat('abc', 1), repeat('abc', 2), repeat('ab', 5);
+SELECT repeat('abc', 0), repeat('', 5), repeat('abc', -1);
+SELECT repeat('abc', 10000) = (SELECT string_agg('abc', '') FROM generate_series(1, 10000));
+
 SELECT ltrim('zzzytrim', 'xyz');
 
 SELECT translate('', '14', 'ax');
-- 
2.55.0

From 6dcad3ba7bd74e8799a4cf8a129f1918d49aa20d Mon Sep 17 00:00:00 2001
From: Sehrope Sarkuni <[email protected]>
Date: Thu, 24 Sep 2026 11:55:30 +0000
Subject: [PATCH v5 2/3] Copy whole repetitions of the pad string at once in
 lpad() and rpad()

append_padding() called pg_mblen_range() and memcpy() once per
character.  Walk the pad string once to count its characters, write
the whole repetitions with a new repeat_bytes() helper, and copy the
partial final repetition as bytes.  The pad string is validated as far
as before, which is all of it when a full repetition is copied and
otherwise only the characters copied.

repeat_bytes() copies the string once, doubles the copied region
until it is at least 16 kB, and then copies that region repeatedly,
checking for interrupts between copies.  Doubling needs only a few
memcpy() calls for short strings, and the fixed block stays in cache
for large results, which pure doubling did not.  This also makes long
lpad() and rpad() calls cancellable, which they were not before.

Co-authored-by: Nathan Bossart <[email protected]>
Discussion: CAH7T-apj+pFg9bRkXVGfGejS2Uu4WzdMd7KoLKTW11mdsrt1Ew@mail.gmail.com">https://postgr.es/m/CAH7T-apj+pFg9bRkXVGfGejS2Uu4WzdMd7KoLKTW11mdsrt1Ew@mail.gmail.com
---
 src/backend/utils/adt/oracle_compat.c  | 93 +++++++++++++++++++++++---
 src/test/regress/expected/encoding.out | 37 ++++++++++
 src/test/regress/expected/strings.out  | 33 +++++++++
 src/test/regress/sql/encoding.sql      | 10 +++
 src/test/regress/sql/strings.sql       |  9 +++
 5 files changed, 174 insertions(+), 8 deletions(-)

diff --git a/src/backend/utils/adt/oracle_compat.c b/src/backend/utils/adt/oracle_compat.c
index e5238e44813..6fdbd595a9f 100644
--- a/src/backend/utils/adt/oracle_compat.c
+++ b/src/backend/utils/adt/oracle_compat.c
@@ -143,29 +143,106 @@ casefold(PG_FUNCTION_ARGS)
 }
 
 
+/*
+ * Write count copies of the srclen-byte string src to dst and return a
+ * pointer past the last byte written.
+ *
+ * The first copy is written with memcpy(), then the region written so far
+ * is copied onto its own end, doubling it each time, until it is at least
+ * REPEAT_BYTES_BLOCK long.  From there on that region is copied repeatedly.
+ * The doubling keeps the number of memcpy() calls small for short strings,
+ * and the fixed block, which is a whole number of copies and stays in
+ * cache, keeps the source reads cheap for large outputs.  The caller must
+ * have checked that count * srclen bytes fit in dst.
+ */
+#define REPEAT_BYTES_BLOCK	16384
+
+static char *
+repeat_bytes(char *dst, const char *src, int srclen, int count)
+{
+	Size		total = (Size) srclen * count;
+	Size		written;
+	Size		block;
+
+	if (total == 0)
+		return dst;
+	Assert(srclen > 0 && count > 0);
+
+	memcpy(dst, src, srclen);
+	written = srclen;
+	while (written < total && written < REPEAT_BYTES_BLOCK)
+	{
+		Size		n = Min(written, total - written);
+
+		memcpy(dst + written, dst, n);
+		written += n;
+	}
+
+	block = written;
+	while (written < total)
+	{
+		Size		n = Min(block, total - written);
+
+		memcpy(dst + written, dst, n);
+		written += n;
+		CHECK_FOR_INTERRUPTS();
+	}
+
+	return dst + total;
+}
+
 /*
  * Append m characters of the padding string pad (padlen bytes) to dst,
  * cycling through pad as needed, and return a pointer past the last byte
  * written.
+ *
+ * The pad string is validated with pg_mblen_range() only as far as it is
+ * used, so an incomplete multibyte character at its end is an error only
+ * if the padding reaches it.
  */
 static char *
 append_padding(char *dst, const char *pad, int padlen, int m)
 {
 	const char *p = pad;
 	const char *pend = pad + padlen;
+	int			nchars = 0;
+	int			nrep;
+	int			tail;
+
+	Assert(padlen > 0 || m <= 0);
+
+	if (m <= 0 || padlen <= 0)
+		return dst;
 
-	while (m--)
+	/* count the characters of one repetition, stopping at m */
+	while (p < pend && nchars < m)
 	{
-		int			mlen = pg_mblen_range(p, pend);
+		p += pg_mblen_range(p, pend);
+		nchars++;
+	}
 
-		memcpy(dst, p, mlen);
-		dst += mlen;
-		p += mlen;
-		if (p == pend)			/* wrap around at end of pad */
-			p = pad;
+	/* whole repetitions only if the pad string was counted to its end */
+	nrep = (p == pend) ? m / nchars : 0;
+	if (nrep == 0)
+	{
+		/* the loop above stopped after exactly m characters */
+		memcpy(dst, pad, p - pad);
+		return dst + (p - pad);
 	}
 
-	return dst;
+	dst = repeat_bytes(dst, pad, padlen, nrep);
+
+	/*
+	 * The remaining m - nrep * nchars characters are a prefix of pad that the
+	 * loop above already checked, so measure them with the unbounded variant
+	 * and copy them as bytes.
+	 */
+	p = pad;
+	for (tail = m - nrep * nchars; tail > 0; tail--)
+		p += pg_mblen_unbounded(p);
+	memcpy(dst, pad, p - pad);
+
+	return dst + (p - pad);
 }
 
 /********************************************************************
diff --git a/src/test/regress/expected/encoding.out b/src/test/regress/expected/encoding.out
index 0bb72a1df6f..9fc871215b0 100644
--- a/src/test/regress/expected/encoding.out
+++ b/src/test/regress/expected/encoding.out
@@ -60,6 +60,43 @@ SELECT reverse(good) FROM regress_encoding;
  éfac
 (1 row)
 
+-- multibyte pad strings: whole repetitions, a partial final repetition, and
+-- fewer than one repetition
+SELECT lpad(good, 7, 'é'), rpad(good, 7, 'é') FROM regress_encoding;
+  lpad   |  rpad   
+---------+---------
+ ééécafé | caféééé
+(1 row)
+
+SELECT lpad(good, 12, 'éab'), rpad(good, 12, 'éab') FROM regress_encoding;
+     lpad     |     rpad     
+--------------+--------------
+ éabéabéacafé | cafééabéabéa
+(1 row)
+
+SELECT lpad(good, 5, 'éab'), rpad(good, 5, 'éab') FROM regress_encoding;
+ lpad  | rpad  
+-------+-------
+ écafé | caféé
+(1 row)
+
+-- a lone lead byte in the pad string is an error if the padding reaches it
+SELECT lpad(good, 7, test_bytea_to_text('\xc3')) FROM regress_encoding;
+ERROR:  invalid byte sequence for encoding "UTF8": 0xc3
+SELECT rpad(good, 7, 'ab' || test_bytea_to_text('\xc3')) FROM regress_encoding;
+ERROR:  invalid byte sequence for encoding "UTF8": 0xc3
+SELECT lpad(good, 5, 'ab' || test_bytea_to_text('\xc3')), rpad(good, 6, 'ab' || test_bytea_to_text('\xc3')) FROM regress_encoding;
+ lpad  |  rpad  
+-------+--------
+ acafé | caféab
+(1 row)
+
+SELECT lpad(good, 4, test_bytea_to_text('\xc3')), rpad(good, 4, test_bytea_to_text('\xc3')) FROM regress_encoding;
+ lpad | rpad 
+------+------
+ café | café
+(1 row)
+
 -- invalid short mb character = error
 SELECT length(truncated) FROM regress_encoding;
 ERROR:  invalid byte sequence for encoding "UTF8": 0xc3
diff --git a/src/test/regress/expected/strings.out b/src/test/regress/expected/strings.out
index fa29abfd829..b588398d7fe 100644
--- a/src/test/regress/expected/strings.out
+++ b/src/test/regress/expected/strings.out
@@ -3441,6 +3441,39 @@ SELECT rpad('hi', 5, '');
  hi
 (1 row)
 
+-- whole repetitions of the pad string, a partial final repetition, and
+-- fewer than one repetition
+SELECT lpad('hi', 8, 'abc'), rpad('hi', 8, 'abc');
+   lpad   |   rpad   
+----------+----------
+ abcabchi | hiabcabc
+(1 row)
+
+SELECT lpad('hi', 9, 'abc'), rpad('hi', 9, 'abc');
+   lpad    |   rpad    
+-----------+-----------
+ abcabcahi | hiabcabca
+(1 row)
+
+SELECT lpad('hi', 12, 'ab'), rpad('hi', 12, 'ab');
+     lpad     |     rpad     
+--------------+--------------
+ abababababhi | hiababababab
+(1 row)
+
+SELECT lpad('hi', 3, 'abc'), rpad('hi', 3, 'abc');
+ lpad | rpad 
+------+------
+ ahi  | hia
+(1 row)
+
+-- padding longer than the 16 kB block that repeat_bytes() copies
+SELECT rpad('', 50000, 'abc') = (SELECT string_agg('abc', '') FROM generate_series(1, 16666)) || 'ab';
+ ?column? 
+----------
+ t
+(1 row)
+
 SELECT ltrim('zzzytrim', 'xyz');
  ltrim 
 -------
diff --git a/src/test/regress/sql/encoding.sql b/src/test/regress/sql/encoding.sql
index 26caa93a5d5..3ea6e54e52d 100644
--- a/src/test/regress/sql/encoding.sql
+++ b/src/test/regress/sql/encoding.sql
@@ -37,6 +37,16 @@ SELECT substring(good, 3, 1) FROM regress_encoding;
 SELECT substring(good, 4, 1) FROM regress_encoding;
 SELECT regexp_replace(good, '^caf(.)$', '\1') FROM regress_encoding;
 SELECT reverse(good) FROM regress_encoding;
+-- multibyte pad strings: whole repetitions, a partial final repetition, and
+-- fewer than one repetition
+SELECT lpad(good, 7, 'é'), rpad(good, 7, 'é') FROM regress_encoding;
+SELECT lpad(good, 12, 'éab'), rpad(good, 12, 'éab') FROM regress_encoding;
+SELECT lpad(good, 5, 'éab'), rpad(good, 5, 'éab') FROM regress_encoding;
+-- a lone lead byte in the pad string is an error if the padding reaches it
+SELECT lpad(good, 7, test_bytea_to_text('\xc3')) FROM regress_encoding;
+SELECT rpad(good, 7, 'ab' || test_bytea_to_text('\xc3')) FROM regress_encoding;
+SELECT lpad(good, 5, 'ab' || test_bytea_to_text('\xc3')), rpad(good, 6, 'ab' || test_bytea_to_text('\xc3')) FROM regress_encoding;
+SELECT lpad(good, 4, test_bytea_to_text('\xc3')), rpad(good, 4, test_bytea_to_text('\xc3')) FROM regress_encoding;
 
 -- invalid short mb character = error
 SELECT length(truncated) FROM regress_encoding;
diff --git a/src/test/regress/sql/strings.sql b/src/test/regress/sql/strings.sql
index 7d9c7275a02..5db0df1c6ae 100644
--- a/src/test/regress/sql/strings.sql
+++ b/src/test/regress/sql/strings.sql
@@ -1165,6 +1165,15 @@ SELECT rpad('hi', -5, 'xy');
 SELECT rpad('hello', 2);
 SELECT rpad('hi', 5, '');
 
+-- whole repetitions of the pad string, a partial final repetition, and
+-- fewer than one repetition
+SELECT lpad('hi', 8, 'abc'), rpad('hi', 8, 'abc');
+SELECT lpad('hi', 9, 'abc'), rpad('hi', 9, 'abc');
+SELECT lpad('hi', 12, 'ab'), rpad('hi', 12, 'ab');
+SELECT lpad('hi', 3, 'abc'), rpad('hi', 3, 'abc');
+-- padding longer than the 16 kB block that repeat_bytes() copies
+SELECT rpad('', 50000, 'abc') = (SELECT string_agg('abc', '') FROM generate_series(1, 16666)) || 'ab';
+
 SELECT ltrim('zzzytrim', 'xyz');
 
 SELECT translate('', '14', 'ax');
-- 
2.55.0

Reply via email to