Performance was tested with GCC 16 on an i7-5600U with
CFLAGS='-march=native -O3 -flto' with the following setup:

$ yes | head -n10M > sl.in
$ yes $(yes eeeaae | head -n10K | paste -s -d' ') | head -n10K > llw.in
$ yes $(yes eeeaae | head -n9 | paste -s -d' ') | head -n1M > asw.in
$ yes $(yes éééááé | head -n9 | paste -s -d$'\xe2\x80\x83') |
   head -n1M > mbw.in

for type in sl llw asw mbw; do
  cat $type.in >/dev/null;
  for imp in '-before' '-after'; do
    echo ============ "${imp:-base}" $type ==============;
    for d in -c; do
      fields='-f1 -f10 -f100'
      test "$d" = "-c" && { fields='-c1 -c10 -c100000000'; d=''; }
      for f in $fields; do
        for loc in C.UTF-8; do
          # Skip -b for UTF-8 as no different
          test "$loc" = C.UTF-8 && echo "$f" | grep -q -- -b \
           && continue
          # Skip multi-byte delimiter for C and not allowed
          test "$loc" = C && test $(echo -n "$d" | wc -c) -ge 4 \
           && continue
          bin=src/cut${imp}
          LC_ALL=$loc $bin $f $d /dev/null 2>/dev/null &&
          hyperfine --warmup=2 -m2 -M6 \
           "LC_ALL=$loc $bin $f $d $type.in >/dev/null" ||
          printf 'Benchmark 1: %s\n  unsupported\n\n' \
           "LC_ALL=$loc $bin $f $d $type.in >/dev/null"
        done;
      done;
    done;
  done;
done

After a little post-processing of the results, we get:

-- src/cut-before

| command          |       sl |      llw |      asw |      mbw |
| ---------------- | -------- | -------- | -------- | -------- |
| UTF8 -c1         |  60.7 ms |  1.244 s | 123.5 ms | 515.2 ms |
| UTF8 -c10        |  39.3 ms |  1.248 s | 127.4 ms | 526.0 ms |
| UTF8 -c100000000 |  39.4 ms |  1.242 s | 121.4 ms | 514.3 ms |

-- src/cut-after

| command          |       sl |      llw |      asw |      mbw |
| ---------------- | -------- | -------- | -------- | -------- |
| UTF8 -c1         |  70.9 ms | 105.5 ms |  27.0 ms |  47.6 ms |
| UTF8 -c10        |  55.7 ms | 105.7 ms |  48.2 ms | 126.8 ms |
| UTF8 -c100000000 |  55.2 ms |  1.596 s | 158.8 ms | 555.1 ms |

* src/cut.c (cut_characters_mode): Shortcut the line delimiter
search if finished processing characters.
* tests/cut/cut.pl: Add a test case.
---
 src/cut.c        | 11 +++++++++++
 tests/cut/cut.pl |  4 ++++
 2 files changed, 15 insertions(+)

diff --git a/src/cut.c b/src/cut.c
index 59b213f39..248e19a14 100644
--- a/src/cut.c
+++ b/src/cut.c
@@ -907,6 +907,17 @@ cut_characters_mode (FILE *stream, bool byte_mode)
 
   while (true)
     {
+      /* Skip unselected line suffixes without decoding each character.  */
+      if (idx && field_selection_exhausted (idx))
+        {
+          idx_t available = mbbuf_topup (&mbbuf);
+          char *p = mbbuf.buffer + mbbuf.offset;
+          char *end = search_bytes (p, line_delim, available);
+          mbbuf_advance (&mbbuf, end ? end - p : available);
+          if (!end && available)
+            continue;
+        }
+
       mcel_t g = mbbuf_get_char (&mbbuf);
 
       if (g.ch == line_delim)
diff --git a/tests/cut/cut.pl b/tests/cut/cut.pl
index c0c962ea0..da4dd0740 100755
--- a/tests/cut/cut.pl
+++ b/tests/cut/cut.pl
@@ -307,6 +307,10 @@ if ($mb_locale ne 'C')
        {ENV => "LC_ALL=$mb_locale"}],
       ['mb-char-5', '-c1-2', {IN=>"\xc3x\n"}, {OUT=>"\xc3x\n"},
        {ENV => "LC_ALL=$mb_locale"}],
+      # Skip unselected suffixes across input buffers and reset at line ends.
+      ['mb-char-suffix', '-c1',
+       {IN=>"\xc3\xa9" . ("\xff" x (2 * $IO_BUFSIZE)) . "\nbignored"},
+       {OUT=>"\xc3\xa9\nb\n"}, {ENV => "LC_ALL=$mb_locale"}],
       # Note mb-byte-n-1 and mb-byte-n-4 differ from coreutils-i18n patch,
       # which outputs a character if any byte is selected.
       # I.e., the i18n patch may output more bytes that the requested range.
-- 
2.55.0


Reply via email to