These are the most common input encodings so worth optimizing for.

Performance was tested with GCC 16 on an i7-5600U with
CFLAGS='-march=native -O3 -flto' with the following setup:

$ yes | head -n10M > sl.in
$ yes $(yes eeeaae | head -n10K | paste -s -d' ') | head -n10K > llw.in
$ yes $(yes eeeaae | head -n9 | paste -s -d' ') | head -n1M > asw.in
$ yes $(yes éééááé | head -n9 | paste -s -d$'\xe2\x80\x83') |
   head -n1M > mbw.in

for type in sl llw asw mbw; do
  cat $type.in >/dev/null;
  for imp in '-before' '-after'; do
    echo ============ "${imp:-base}" $type ==============;
    for d in -c; do
      fields='-f1 -f10 -f100'
      test "$d" = "-c" && { fields='-c1 -c10 -c100000000'; d=''; }
      for f in $fields; do
        for loc in C.UTF-8; do
          # Skip -b for UTF-8 as no different
          test "$loc" = C.UTF-8 && echo "$f" | grep -q -- -b \
           && continue
          # Skip multi-byte delimiter for C and not allowed
          test "$loc" = C && test $(echo -n "$d" | wc -c) -ge 4 \
           && continue
          bin=src/cut${imp}
          LC_ALL=$loc $bin $f $d /dev/null 2>/dev/null &&
          hyperfine --warmup=2 -m2 -M6 \
           "LC_ALL=$loc $bin $f $d $type.in >/dev/null" ||
          printf 'Benchmark 1: %s\n  unsupported\n\n' \
           "LC_ALL=$loc $bin $f $d $type.in >/dev/null"
        done;
      done;
    done;
  done;
done

After a little post-processing of the results, we get:

-- src/cut-before

| command         |       sl |      llw |      asw |      mbw |
| --------------- | -------- | -------- | -------- | -------- |
| UTF8 -c1        |  70.7 ms | 103.5 ms |  26.9 ms |  48.6 ms |
| UTF8 -c10       |  55.1 ms | 103.9 ms |  48.6 ms | 127.3 ms |
| UTF8 -c100000   |  54.7 ms |  1.585 s | 156.9 ms | 555.9 ms |

-- src/cut-after

| command         |       sl |      llw |      asw |      mbw |
| --------------- | -------- | -------- | -------- | -------- |
| UTF8 -c1        |  90.1 ms | 103.6 ms |  28.0 ms |  50.9 ms |
| UTF8 -c10       |  39.7 ms | 103.2 ms |  40.0 ms | 127.5 ms |
| UTF8 -c100000   |  39.0 ms | 104.4 ms |  18.5 ms |  30.3 ms |

* bootstrap.conf: Explicitly depend on unistr/u8-check.
* src/cut.c (cut_characters_mode): Quickly skip over unselected
ASCII/UTF-8 data.
* tests/cut/cut.pl: Add test cases.
---
 bootstrap.conf   |  1 +
 src/cut.c        | 37 +++++++++++++++++++++++++++++++++++++
 tests/cut/cut.pl | 15 +++++++++++++++
 3 files changed, 53 insertions(+)

diff --git a/bootstrap.conf b/bootstrap.conf
index 7fd4a9cf4..d2bea314e 100644
--- a/bootstrap.conf
+++ b/bootstrap.conf
@@ -310,6 +310,7 @@ gnulib_modules="
   uname
   unicodeio
   unistd-safer
+  unistr/u8-check
   unlink-busy
   unlinkat
   unlinkdir
diff --git a/src/cut.c b/src/cut.c
index 248e19a14..2e6d8bf82 100644
--- a/src/cut.c
+++ b/src/cut.c
@@ -38,6 +38,7 @@
 #include "memchr2.h"
 
 #include "set-fields.h"
+#include "unistr.h"
 
 /* The official name of this program (e.g., no 'g' prefix).  */
 #define PROGRAM_NAME "cut"
@@ -897,6 +898,7 @@ cut_bytes (FILE *stream)
 static void
 cut_characters_mode (FILE *stream, bool byte_mode)
 {
+  bool utf8 = is_utf8_charset ();
   uintmax_t idx = 0;
   bool print_delimiter = false;
   static char bytes_in[IO_BUFSIZE];
@@ -918,6 +920,41 @@ cut_characters_mode (FILE *stream, bool byte_mode)
             continue;
         }
 
+      /* Count unselected ASCII/UTF-8 characters directly, for efficiency.  */
+      if (!byte_mode && utf8 && idx + 1 < current_rp->lo)
+        {
+          char const *p = mbbuf.buffer + mbbuf.offset;
+          idx_t n = MIN (mbbuf_avail (&mbbuf), current_rp->lo - idx - 1);
+          char const *end = search_bytes (p, line_delim, n);
+          if (end)
+            {
+              /* Even with ASCII, this cannot reach the next selection.  */
+              mbbuf_advance (&mbbuf, end - p + 1);
+              reset_item_line (&idx, &print_delimiter);
+              continue;
+            }
+
+          /* Detect non ASCII.  */
+          unsigned char bits = n ? p[0] : 0;
+          if (!(bits & 0x80))
+            for (idx_t i = 0; i < n; i++)
+              bits |= p[i];
+          uintmax_t count = n;
+          if (bits & 0x80) /* any UTF8 */
+            {
+              uint8_t const *invalid = u8_check ((uint8_t const *) p, n);
+              if (invalid)
+                n = (char const *) invalid - p;
+              /* Count each non-continuation byte, i.e., each utf8 char.  */
+              count = 0;
+              for (idx_t i = 0; i < n; i++)
+                count += (to_uchar (p[i]) & 0xc0) != 0x80;
+            }
+
+          idx += count;
+          mbbuf_advance (&mbbuf, n);
+        }
+
       mcel_t g = mbbuf_get_char (&mbbuf);
 
       if (g.ch == line_delim)
diff --git a/tests/cut/cut.pl b/tests/cut/cut.pl
index da4dd0740..2ea4d52c7 100755
--- a/tests/cut/cut.pl
+++ b/tests/cut/cut.pl
@@ -307,6 +307,21 @@ if ($mb_locale ne 'C')
        {ENV => "LC_ALL=$mb_locale"}],
       ['mb-char-5', '-c1-2', {IN=>"\xc3x\n"}, {OUT=>"\xc3x\n"},
        {ENV => "LC_ALL=$mb_locale"}],
+      # Skip lines whose remaining bytes cannot reach the next selection.
+      ['mb-char-short-line', '-c4',
+       {IN=>"\xc3\xa9\n\nabcX\nab\xffY\nz"}, {OUT=>"\n\nX\nY\n\n"},
+       {ENV => "LC_ALL=$mb_locale"}],
+      # Count multibyte prefixes across buffers, including invalid bytes.
+      ['mb-char-prefix-ascii', '-c' . $IO_BUFSIZE,
+       {IN=>("a" x ($IO_BUFSIZE - 2)) . "\xc3\xa9X"},
+       {OUT=>"X\n"}, {ENV => "LC_ALL=$mb_locale"}],
+      ['mb-char-prefix', '-c' . ($IO_BUFSIZE + 4),
+       {IN=>("\xc3\xa9" x $IO_BUFSIZE)
+            . "\xe2\x82\xac\xf0\x9f\x98\x80\xffX\nshort"},
+       {OUT=>"X\n\n"}, {ENV => "LC_ALL=$mb_locale"}],
+      ['mb-char-prefix-nul', '-z', '-c5',
+       {IN=>"\xc3\xa9\xe2\x82\xac\xf0\x9f\x98\x80\xffX\0short"},
+       {OUT=>"X\0t\0"}, {ENV => "LC_ALL=$mb_locale"}],
       # Skip unselected suffixes across input buffers and reset at line ends.
       ['mb-char-suffix', '-c1',
        {IN=>"\xc3\xa9" . ("\xff" x (2 * $IO_BUFSIZE)) . "\nbignored"},
-- 
2.55.0


Reply via email to