Generate ASCII and UTF-8 data:
  $ yes $(yes eeeaae | head -n10K | paste -s -d,) | head -n10K > ll.in
  $ yes $(yes éééááé | head -n9 | paste -s -d,) | head -n1M > mb.in

ASCII is 14x faster:
  $ time src/wc-before -m < ll.in
  real  0m1.801s
  $ time src/wc-after -m < ll.in
  real  0m0.123s

UTF-8 is 7x faster:
  $ time src/wc-before -m < mb.in
  real  0m1.399s
  $ time src/wc-after -m < mb.in
  real  0m0.189s

* src/wc.c (wc): Use the u8_count() routine like cut(1),
to optimize counting valid UTF-8 characters.
* tests/wc/wc.pl: Add a test case.
* NEWS: Mention the improvement.
---
 NEWS           |  2 ++
 src/wc.c       | 16 ++++++++++++++++
 tests/wc/wc.pl |  6 ++++++
 3 files changed, 24 insertions(+)

diff --git a/NEWS b/NEWS
index 6daa691ad..6f54390a7 100644
--- a/NEWS
+++ b/NEWS
@@ -36,6 +36,8 @@ GNU coreutils NEWS                                    -*- 
outline -*-
   initial reads finish, but subsequent reads might block, which was seen
   e.g. with fifos on NetBSD 10.
 
+  'wc -m' counts ASCII up to 14x faster, and UTF-8 up to 7x faster.
+
 ** Build-related
 
   gzip-compressed tarballs are no longer distributed,
diff --git a/src/wc.c b/src/wc.c
index 3bf6dcbc4..272a905f5 100644
--- a/src/wc.c
+++ b/src/wc.c
@@ -41,6 +41,7 @@
 #include "cpu-supports.h"
 #include "ioblksize.h"
 #include "wc.h"
+#include "utf8.h"
 
 /* The official name of this program (e.g., no 'g' prefix).  */
 #define PROGRAM_NAME "wc"
@@ -479,6 +480,8 @@ wc (int fd, char const *file_x, struct fstatus *fstatus)
       intmax_t linepos = 0;
       mbstate_t state; mbszero (&state);
       bool in_shift = false;
+      bool count_utf8 = !count_complicated && !print_lines
+                        && is_utf8_charset ();
       idx_t prev = 0; /* Number of bytes carried over from previous round.  */
 
       for (ssize_t bytes_read;
@@ -495,8 +498,21 @@ wc (int fd, char const *file_x, struct fstatus *fstatus)
           bytes += bytes_read;
           char const *p = buf;
           char const *plim = p + prev + bytes_read;
+          bool try_count_utf8 = count_utf8;
           do
             {
+              if (try_count_utf8 && !in_shift)
+                {
+                  /* Scan at most once per buffer, to avoid repeatedly
+                     scanning ASCII after decoding errors.  */
+                  try_count_utf8 = false;
+                  idx_t n = plim - p;
+                  chars += u8_count (p, &n);
+                  p += n;
+                  if (p == plim)
+                    break;
+                }
+
               char32_t wide_char;
               idx_t charbytes;
               bool single_byte;
diff --git a/tests/wc/wc.pl b/tests/wc/wc.pl
index a1b25b0b8..49e9006c8 100755
--- a/tests/wc/wc.pl
+++ b/tests/wc/wc.pl
@@ -112,6 +112,12 @@ if (defined $mb_locale && $mb_locale ne 'none')
       ['mb-split-space', '-w', '<',
        {IN=>('a' x ($bufsize - 1)) . "\xe2\x80\x83"},
        {OUT=>"1\n"}, {ENV=>"LC_ALL=$mb_locale"}];
+
+    # Ensure we handle a split character, NUL, and encoding errors.
+    push @Tests,
+      ['mb-count-buffer', '-m', '<',
+       {IN=>('a' x ($bufsize - 1)) . "\xc3\xa9\0\xffZ\xe2\x82"},
+       {OUT=>($bufsize + 2) . "\n"}, {ENV=>"LC_ALL=$mb_locale"}];
   }
 
 my $save_temps = $ENV{DEBUG};
-- 
2.55.0


Reply via email to