Commit v9.4-43-ga6064bb86 introduced this edge case
in handling multi-byte characters spanning the input buffer.

* src/wc.c (wc): Ensure we include the bytes previously read
but as yet unprocessed.
* tests/wc/mb-non-utf8.sh: Add a new test for miscounted GB118030 chars.
* tests/local.mk: Reference the new test.
* tests/wc/wc.pl: Add a new check for word count with a
UTF-8 space spanning a buffer boundary.
* NEWS: Mention the bug fix.
---
 NEWS                    |  5 +++++
 src/wc.c                |  5 ++++-
 tests/local.mk          |  1 +
 tests/wc/mb-non-utf8.sh | 36 ++++++++++++++++++++++++++++++++++++
 tests/wc/wc.pl          | 11 +++++++++++
 5 files changed, 57 insertions(+), 1 deletion(-)
 create mode 100755 tests/wc/mb-non-utf8.sh

diff --git a/NEWS b/NEWS
index ab6f7beb3..6daa691ad 100644
--- a/NEWS
+++ b/NEWS
@@ -9,6 +9,11 @@ GNU coreutils NEWS                                    -*- 
outline -*-
   resolves to a path beginning with the symbolic link itself.
   [bug introduced in coreutils-9.0]
 
+  'wc' no longer miscounts characters or words when a multi-byte character
+  spans input buffers.  E.g. this could affect character counts
+  in GB18030 locales, and word counts in UTF-8 locales.
+  [bug introduced in coreutils-9.5]
+
 ** New Features
 
   'env' and 'printenv' now support the --quoting-style option
diff --git a/src/wc.c b/src/wc.c
index 057b710d6..3bf6dcbc4 100644
--- a/src/wc.c
+++ b/src/wc.c
@@ -511,6 +511,7 @@ wc (int fd, char const *file_x, struct fstatus *fstatus)
                 }
               else
                 {
+                  idx_t prev_bytes = prev;
                   idx_t scanbytes = plim - (p + prev);
                   size_t n = mbrtoc32 (&wide_char, p + prev, scanbytes, 
&state);
                   prev = 0;
@@ -550,7 +551,9 @@ wc (int fd, char const *file_x, struct fstatus *fstatus)
                       continue;
                     }
 
-                  charbytes = n + !n;
+                  /* Include bytes already consumed into STATE in an earlier
+                     read, but still present at P in the buffer.  */
+                  charbytes = prev_bytes + n + !n;
                   single_byte = charbytes == !in_shift;
                   in_shift = !mbsinit (&state);
                 }
diff --git a/tests/local.mk b/tests/local.mk
index ca2a0b8b7..dbf5789df 100644
--- a/tests/local.mk
+++ b/tests/local.mk
@@ -309,6 +309,7 @@ all_tests =                                 \
   tests/cut/mb-non-utf8.sh                     \
   tests/cut/bounded-memory.sh                  \
   tests/cut/cut-huge-range.sh                  \
+  tests/wc/mb-non-utf8.sh                      \
   tests/wc/wc.pl                               \
   tests/wc/wc-cpu.sh                           \
   tests/wc/wc-files0-from.pl                   \
diff --git a/tests/wc/mb-non-utf8.sh b/tests/wc/mb-non-utf8.sh
new file mode 100755
index 000000000..e5f1babad
--- /dev/null
+++ b/tests/wc/mb-non-utf8.sh
@@ -0,0 +1,36 @@
+#!/bin/sh
+# Count GB18030 characters spanning read buffers.
+
+# Copyright (C) 2026 Free Software Foundation, Inc.
+
+# This program is free software: you can redistribute it and/or modify
+# it under the terms of the GNU General Public License as published by
+# the Free Software Foundation, either version 3 of the License, or
+# (at your option) any later version.
+
+# This program is distributed in the hope that it will be useful,
+# but WITHOUT ANY WARRANTY; without even the implied warranty of
+# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
+# GNU General Public License for more details.
+
+# You should have received a copy of the GNU General Public License
+# along with this program.  If not, see <https://www.gnu.org/licenses/>.
+
+. "${srcdir=.}/tests/init.sh"; path_prepend_ ./src
+print_ver_ wc printf
+getlimits_
+
+export LC_ALL=zh_CN.gb18030
+test "$(locale charmap 2>/dev/null | sed 's/gb/GB/')" = GB18030 ||
+  skip_ 'GB18030 charset support not detected'
+
+# U+10000 is encoded as 90 30 81 30.  Its ASCII bytes must not be
+# counted again when the character is split across read buffers.
+padding=$(($IO_BUFSIZE - 1))
+head -c "$padding" /dev/zero | tr '\000' a > in || framework_failure_
+env printf '\x90\x30\x81\x30' >> in || framework_failure_
+printf '%s\n' "$IO_BUFSIZE" > exp || framework_failure_
+wc -m < in > out || fail=1
+compare exp out || fail=1
+
+Exit $fail
diff --git a/tests/wc/wc.pl b/tests/wc/wc.pl
index b57cd5145..a1b25b0b8 100755
--- a/tests/wc/wc.pl
+++ b/tests/wc/wc.pl
@@ -103,6 +103,17 @@ if (defined $single_byte_locale)
     push @Tests, @new;
   };
 
+# A split EM SPACE must not leave continuation bytes to start another word.
+# Separated here so we read from file and thus exercise IO_BUFSIZE.
+if (defined $mb_locale && $mb_locale ne 'none')
+  {
+    my $bufsize = getlimits ()->{IO_BUFSIZE};
+    push @Tests,
+      ['mb-split-space', '-w', '<',
+       {IN=>('a' x ($bufsize - 1)) . "\xe2\x80\x83"},
+       {OUT=>"1\n"}, {ENV=>"LC_ALL=$mb_locale"}];
+  }
+
 my $save_temps = $ENV{DEBUG};
 my $verbose = $ENV{VERBOSE};
 
-- 
2.55.0


Reply via email to