These are the most common input encodings so worth optimizing for.
Performance was tested with GCC 16 on an i7-5600U with
CFLAGS='-march=native -O3 -flto' with the following setup:
$ yes | head -n10M > sl.in
$ yes $(yes eeeaae | head -n10K | paste -s -d' ') | head -n10K > llw.in
$ yes $(yes eeeaae | head -n9 | paste -s -d' ') | head -n1M > asw.in
$ yes $(yes éééááé | head -n9 | paste -s -d$'\xe2\x80\x83') |
head -n1M > mbw.in
for type in sl llw asw mbw; do
cat $type.in >/dev/null;
for imp in '-before' '-after'; do
echo ============ "${imp:-base}" $type ==============;
for d in -c; do
fields='-f1 -f10 -f100'
test "$d" = "-c" && { fields='-c1 -c10 -c100000000'; d=''; }
for f in $fields; do
for loc in C.UTF-8; do
# Skip -b for UTF-8 as no different
test "$loc" = C.UTF-8 && echo "$f" | grep -q -- -b \
&& continue
# Skip multi-byte delimiter for C and not allowed
test "$loc" = C && test $(echo -n "$d" | wc -c) -ge 4 \
&& continue
bin=src/cut${imp}
LC_ALL=$loc $bin $f $d /dev/null 2>/dev/null &&
hyperfine --warmup=2 -m2 -M6 \
"LC_ALL=$loc $bin $f $d $type.in >/dev/null" ||
printf 'Benchmark 1: %s\n unsupported\n\n' \
"LC_ALL=$loc $bin $f $d $type.in >/dev/null"
done;
done;
done;
done;
done
After a little post-processing of the results, we get:
-- src/cut-before
| command | sl | llw | asw | mbw |
| --------------- | -------- | -------- | -------- | -------- |
| UTF8 -c1 | 70.7 ms | 103.5 ms | 26.9 ms | 48.6 ms |
| UTF8 -c10 | 55.1 ms | 103.9 ms | 48.6 ms | 127.3 ms |
| UTF8 -c100000 | 54.7 ms | 1.585 s | 156.9 ms | 555.9 ms |
-- src/cut-after
| command | sl | llw | asw | mbw |
| --------------- | -------- | -------- | -------- | -------- |
| UTF8 -c1 | 90.1 ms | 103.6 ms | 28.0 ms | 50.9 ms |
| UTF8 -c10 | 39.7 ms | 103.2 ms | 40.0 ms | 127.5 ms |
| UTF8 -c100000 | 39.0 ms | 104.4 ms | 18.5 ms | 30.3 ms |
* bootstrap.conf: Explicitly depend on unistr/u8-check.
* src/cut.c (cut_characters_mode): Quickly skip over unselected
ASCII/UTF-8 data.
* tests/cut/cut.pl: Add test cases.
---
bootstrap.conf | 1 +
src/cut.c | 37 +++++++++++++++++++++++++++++++++++++
tests/cut/cut.pl | 15 +++++++++++++++
3 files changed, 53 insertions(+)
diff --git a/bootstrap.conf b/bootstrap.conf
index 7fd4a9cf4..d2bea314e 100644
--- a/bootstrap.conf
+++ b/bootstrap.conf
@@ -310,6 +310,7 @@ gnulib_modules="
uname
unicodeio
unistd-safer
+ unistr/u8-check
unlink-busy
unlinkat
unlinkdir
diff --git a/src/cut.c b/src/cut.c
index 248e19a14..2e6d8bf82 100644
--- a/src/cut.c
+++ b/src/cut.c
@@ -38,6 +38,7 @@
#include "memchr2.h"
#include "set-fields.h"
+#include "unistr.h"
/* The official name of this program (e.g., no 'g' prefix). */
#define PROGRAM_NAME "cut"
@@ -897,6 +898,7 @@ cut_bytes (FILE *stream)
static void
cut_characters_mode (FILE *stream, bool byte_mode)
{
+ bool utf8 = is_utf8_charset ();
uintmax_t idx = 0;
bool print_delimiter = false;
static char bytes_in[IO_BUFSIZE];
@@ -918,6 +920,41 @@ cut_characters_mode (FILE *stream, bool byte_mode)
continue;
}
+ /* Count unselected ASCII/UTF-8 characters directly, for efficiency. */
+ if (!byte_mode && utf8 && idx + 1 < current_rp->lo)
+ {
+ char const *p = mbbuf.buffer + mbbuf.offset;
+ idx_t n = MIN (mbbuf_avail (&mbbuf), current_rp->lo - idx - 1);
+ char const *end = search_bytes (p, line_delim, n);
+ if (end)
+ {
+ /* Even with ASCII, this cannot reach the next selection. */
+ mbbuf_advance (&mbbuf, end - p + 1);
+ reset_item_line (&idx, &print_delimiter);
+ continue;
+ }
+
+ /* Detect non ASCII. */
+ unsigned char bits = n ? p[0] : 0;
+ if (!(bits & 0x80))
+ for (idx_t i = 0; i < n; i++)
+ bits |= p[i];
+ uintmax_t count = n;
+ if (bits & 0x80) /* any UTF8 */
+ {
+ uint8_t const *invalid = u8_check ((uint8_t const *) p, n);
+ if (invalid)
+ n = (char const *) invalid - p;
+ /* Count each non-continuation byte, i.e., each utf8 char. */
+ count = 0;
+ for (idx_t i = 0; i < n; i++)
+ count += (to_uchar (p[i]) & 0xc0) != 0x80;
+ }
+
+ idx += count;
+ mbbuf_advance (&mbbuf, n);
+ }
+
mcel_t g = mbbuf_get_char (&mbbuf);
if (g.ch == line_delim)
diff --git a/tests/cut/cut.pl b/tests/cut/cut.pl
index da4dd0740..2ea4d52c7 100755
--- a/tests/cut/cut.pl
+++ b/tests/cut/cut.pl
@@ -307,6 +307,21 @@ if ($mb_locale ne 'C')
{ENV => "LC_ALL=$mb_locale"}],
['mb-char-5', '-c1-2', {IN=>"\xc3x\n"}, {OUT=>"\xc3x\n"},
{ENV => "LC_ALL=$mb_locale"}],
+ # Skip lines whose remaining bytes cannot reach the next selection.
+ ['mb-char-short-line', '-c4',
+ {IN=>"\xc3\xa9\n\nabcX\nab\xffY\nz"}, {OUT=>"\n\nX\nY\n\n"},
+ {ENV => "LC_ALL=$mb_locale"}],
+ # Count multibyte prefixes across buffers, including invalid bytes.
+ ['mb-char-prefix-ascii', '-c' . $IO_BUFSIZE,
+ {IN=>("a" x ($IO_BUFSIZE - 2)) . "\xc3\xa9X"},
+ {OUT=>"X\n"}, {ENV => "LC_ALL=$mb_locale"}],
+ ['mb-char-prefix', '-c' . ($IO_BUFSIZE + 4),
+ {IN=>("\xc3\xa9" x $IO_BUFSIZE)
+ . "\xe2\x82\xac\xf0\x9f\x98\x80\xffX\nshort"},
+ {OUT=>"X\n\n"}, {ENV => "LC_ALL=$mb_locale"}],
+ ['mb-char-prefix-nul', '-z', '-c5',
+ {IN=>"\xc3\xa9\xe2\x82\xac\xf0\x9f\x98\x80\xffX\0short"},
+ {OUT=>"X\0t\0"}, {ENV => "LC_ALL=$mb_locale"}],
# Skip unselected suffixes across input buffers and reset at line ends.
['mb-char-suffix', '-c1',
{IN=>"\xc3\xa9" . ("\xff" x (2 * $IO_BUFSIZE)) . "\nbignored"},
--
2.55.0