Generate ASCII and UTF-8 data:
$ yes $(yes eeeaae | head -n10K | paste -s -d,) | head -n10K > ll.in
$ yes $(yes éééááé | head -n9 | paste -s -d,) | head -n1M > mb.in
ASCII is 14x faster:
$ time src/wc-before -m < ll.in
real 0m1.801s
$ time src/wc-after -m < ll.in
real 0m0.123s
UTF-8 is 7x faster:
$ time src/wc-before -m < mb.in
real 0m1.399s
$ time src/wc-after -m < mb.in
real 0m0.189s
* src/wc.c (wc): Use the u8_count() routine like cut(1),
to optimize counting valid UTF-8 characters.
* tests/wc/wc.pl: Add a test case.
* NEWS: Mention the improvement.
---
NEWS | 2 ++
src/wc.c | 16 ++++++++++++++++
tests/wc/wc.pl | 6 ++++++
3 files changed, 24 insertions(+)
diff --git a/NEWS b/NEWS
index 6daa691ad..6f54390a7 100644
--- a/NEWS
+++ b/NEWS
@@ -36,6 +36,8 @@ GNU coreutils NEWS -*-
outline -*-
initial reads finish, but subsequent reads might block, which was seen
e.g. with fifos on NetBSD 10.
+ 'wc -m' counts ASCII up to 14x faster, and UTF-8 up to 7x faster.
+
** Build-related
gzip-compressed tarballs are no longer distributed,
diff --git a/src/wc.c b/src/wc.c
index 3bf6dcbc4..272a905f5 100644
--- a/src/wc.c
+++ b/src/wc.c
@@ -41,6 +41,7 @@
#include "cpu-supports.h"
#include "ioblksize.h"
#include "wc.h"
+#include "utf8.h"
/* The official name of this program (e.g., no 'g' prefix). */
#define PROGRAM_NAME "wc"
@@ -479,6 +480,8 @@ wc (int fd, char const *file_x, struct fstatus *fstatus)
intmax_t linepos = 0;
mbstate_t state; mbszero (&state);
bool in_shift = false;
+ bool count_utf8 = !count_complicated && !print_lines
+ && is_utf8_charset ();
idx_t prev = 0; /* Number of bytes carried over from previous round. */
for (ssize_t bytes_read;
@@ -495,8 +498,21 @@ wc (int fd, char const *file_x, struct fstatus *fstatus)
bytes += bytes_read;
char const *p = buf;
char const *plim = p + prev + bytes_read;
+ bool try_count_utf8 = count_utf8;
do
{
+ if (try_count_utf8 && !in_shift)
+ {
+ /* Scan at most once per buffer, to avoid repeatedly
+ scanning ASCII after decoding errors. */
+ try_count_utf8 = false;
+ idx_t n = plim - p;
+ chars += u8_count (p, &n);
+ p += n;
+ if (p == plim)
+ break;
+ }
+
char32_t wide_char;
idx_t charbytes;
bool single_byte;
diff --git a/tests/wc/wc.pl b/tests/wc/wc.pl
index a1b25b0b8..49e9006c8 100755
--- a/tests/wc/wc.pl
+++ b/tests/wc/wc.pl
@@ -112,6 +112,12 @@ if (defined $mb_locale && $mb_locale ne 'none')
['mb-split-space', '-w', '<',
{IN=>('a' x ($bufsize - 1)) . "\xe2\x80\x83"},
{OUT=>"1\n"}, {ENV=>"LC_ALL=$mb_locale"}];
+
+ # Ensure we handle a split character, NUL, and encoding errors.
+ push @Tests,
+ ['mb-count-buffer', '-m', '<',
+ {IN=>('a' x ($bufsize - 1)) . "\xc3\xa9\0\xffZ\xe2\x82"},
+ {OUT=>($bufsize + 2) . "\n"}, {ENV=>"LC_ALL=$mb_locale"}];
}
my $save_temps = $ENV{DEBUG};
--
2.55.0