Performance was tested with GCC 16 on an i7-5600U with
CFLAGS='-march=native -O3 -flto' with the following setup:
$ yes | head -n10M > sl.in
$ yes $(yes eeeaae | head -n10K | paste -s -d' ') | head -n10K > llw.in
$ yes $(yes eeeaae | head -n9 | paste -s -d' ') | head -n1M > asw.in
$ yes $(yes éééááé | head -n9 | paste -s -d$'\xe2\x80\x83') |
head -n1M > mbw.in
for type in sl llw asw mbw; do
cat $type.in >/dev/null;
for imp in '-before' '-after'; do
echo ============ "${imp:-base}" $type ==============;
for d in -c; do
fields='-f1 -f10 -f100'
test "$d" = "-c" && { fields='-c1 -c10 -c100000000'; d=''; }
for f in $fields; do
for loc in C.UTF-8; do
# Skip -b for UTF-8 as no different
test "$loc" = C.UTF-8 && echo "$f" | grep -q -- -b \
&& continue
# Skip multi-byte delimiter for C and not allowed
test "$loc" = C && test $(echo -n "$d" | wc -c) -ge 4 \
&& continue
bin=src/cut${imp}
LC_ALL=$loc $bin $f $d /dev/null 2>/dev/null &&
hyperfine --warmup=2 -m2 -M6 \
"LC_ALL=$loc $bin $f $d $type.in >/dev/null" ||
printf 'Benchmark 1: %s\n unsupported\n\n' \
"LC_ALL=$loc $bin $f $d $type.in >/dev/null"
done;
done;
done;
done;
done
After a little post-processing of the results, we get:
-- src/cut-before
| command | sl | llw | asw | mbw |
| ---------------- | -------- | -------- | -------- | -------- |
| UTF8 -c1 | 60.7 ms | 1.244 s | 123.5 ms | 515.2 ms |
| UTF8 -c10 | 39.3 ms | 1.248 s | 127.4 ms | 526.0 ms |
| UTF8 -c100000000 | 39.4 ms | 1.242 s | 121.4 ms | 514.3 ms |
-- src/cut-after
| command | sl | llw | asw | mbw |
| ---------------- | -------- | -------- | -------- | -------- |
| UTF8 -c1 | 70.9 ms | 105.5 ms | 27.0 ms | 47.6 ms |
| UTF8 -c10 | 55.7 ms | 105.7 ms | 48.2 ms | 126.8 ms |
| UTF8 -c100000000 | 55.2 ms | 1.596 s | 158.8 ms | 555.1 ms |
* src/cut.c (cut_characters_mode): Shortcut the line delimiter
search if finished processing characters.
* tests/cut/cut.pl: Add a test case.
---
src/cut.c | 11 +++++++++++
tests/cut/cut.pl | 4 ++++
2 files changed, 15 insertions(+)
diff --git a/src/cut.c b/src/cut.c
index 59b213f39..248e19a14 100644
--- a/src/cut.c
+++ b/src/cut.c
@@ -907,6 +907,17 @@ cut_characters_mode (FILE *stream, bool byte_mode)
while (true)
{
+ /* Skip unselected line suffixes without decoding each character. */
+ if (idx && field_selection_exhausted (idx))
+ {
+ idx_t available = mbbuf_topup (&mbbuf);
+ char *p = mbbuf.buffer + mbbuf.offset;
+ char *end = search_bytes (p, line_delim, available);
+ mbbuf_advance (&mbbuf, end ? end - p : available);
+ if (!end && available)
+ continue;
+ }
+
mcel_t g = mbbuf_get_char (&mbbuf);
if (g.ch == line_delim)
diff --git a/tests/cut/cut.pl b/tests/cut/cut.pl
index c0c962ea0..da4dd0740 100755
--- a/tests/cut/cut.pl
+++ b/tests/cut/cut.pl
@@ -307,6 +307,10 @@ if ($mb_locale ne 'C')
{ENV => "LC_ALL=$mb_locale"}],
['mb-char-5', '-c1-2', {IN=>"\xc3x\n"}, {OUT=>"\xc3x\n"},
{ENV => "LC_ALL=$mb_locale"}],
+ # Skip unselected suffixes across input buffers and reset at line ends.
+ ['mb-char-suffix', '-c1',
+ {IN=>"\xc3\xa9" . ("\xff" x (2 * $IO_BUFSIZE)) . "\nbignored"},
+ {OUT=>"\xc3\xa9\nb\n"}, {ENV => "LC_ALL=$mb_locale"}],
# Note mb-byte-n-1 and mb-byte-n-4 differ from coreutils-i18n patch,
# which outputs a character if any byte is selected.
# I.e., the i18n patch may output more bytes that the requested range.
--
2.55.0