Commit v9.4-43-ga6064bb86 introduced this edge case
in handling multi-byte characters spanning the input buffer.
* src/wc.c (wc): Ensure we include the bytes previously read
but as yet unprocessed.
* tests/wc/mb-non-utf8.sh: Add a new test for miscounted GB118030 chars.
* tests/local.mk: Reference the new test.
* tests/wc/wc.pl: Add a new check for word count with a
UTF-8 space spanning a buffer boundary.
* NEWS: Mention the bug fix.
---
NEWS | 5 +++++
src/wc.c | 5 ++++-
tests/local.mk | 1 +
tests/wc/mb-non-utf8.sh | 36 ++++++++++++++++++++++++++++++++++++
tests/wc/wc.pl | 11 +++++++++++
5 files changed, 57 insertions(+), 1 deletion(-)
create mode 100755 tests/wc/mb-non-utf8.sh
diff --git a/NEWS b/NEWS
index ab6f7beb3..6daa691ad 100644
--- a/NEWS
+++ b/NEWS
@@ -9,6 +9,11 @@ GNU coreutils NEWS -*-
outline -*-
resolves to a path beginning with the symbolic link itself.
[bug introduced in coreutils-9.0]
+ 'wc' no longer miscounts characters or words when a multi-byte character
+ spans input buffers. E.g. this could affect character counts
+ in GB18030 locales, and word counts in UTF-8 locales.
+ [bug introduced in coreutils-9.5]
+
** New Features
'env' and 'printenv' now support the --quoting-style option
diff --git a/src/wc.c b/src/wc.c
index 057b710d6..3bf6dcbc4 100644
--- a/src/wc.c
+++ b/src/wc.c
@@ -511,6 +511,7 @@ wc (int fd, char const *file_x, struct fstatus *fstatus)
}
else
{
+ idx_t prev_bytes = prev;
idx_t scanbytes = plim - (p + prev);
size_t n = mbrtoc32 (&wide_char, p + prev, scanbytes,
&state);
prev = 0;
@@ -550,7 +551,9 @@ wc (int fd, char const *file_x, struct fstatus *fstatus)
continue;
}
- charbytes = n + !n;
+ /* Include bytes already consumed into STATE in an earlier
+ read, but still present at P in the buffer. */
+ charbytes = prev_bytes + n + !n;
single_byte = charbytes == !in_shift;
in_shift = !mbsinit (&state);
}
diff --git a/tests/local.mk b/tests/local.mk
index ca2a0b8b7..dbf5789df 100644
--- a/tests/local.mk
+++ b/tests/local.mk
@@ -309,6 +309,7 @@ all_tests = \
tests/cut/mb-non-utf8.sh \
tests/cut/bounded-memory.sh \
tests/cut/cut-huge-range.sh \
+ tests/wc/mb-non-utf8.sh \
tests/wc/wc.pl \
tests/wc/wc-cpu.sh \
tests/wc/wc-files0-from.pl \
diff --git a/tests/wc/mb-non-utf8.sh b/tests/wc/mb-non-utf8.sh
new file mode 100755
index 000000000..e5f1babad
--- /dev/null
+++ b/tests/wc/mb-non-utf8.sh
@@ -0,0 +1,36 @@
+#!/bin/sh
+# Count GB18030 characters spanning read buffers.
+
+# Copyright (C) 2026 Free Software Foundation, Inc.
+
+# This program is free software: you can redistribute it and/or modify
+# it under the terms of the GNU General Public License as published by
+# the Free Software Foundation, either version 3 of the License, or
+# (at your option) any later version.
+
+# This program is distributed in the hope that it will be useful,
+# but WITHOUT ANY WARRANTY; without even the implied warranty of
+# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+# GNU General Public License for more details.
+
+# You should have received a copy of the GNU General Public License
+# along with this program. If not, see <https://www.gnu.org/licenses/>.
+
+. "${srcdir=.}/tests/init.sh"; path_prepend_ ./src
+print_ver_ wc printf
+getlimits_
+
+export LC_ALL=zh_CN.gb18030
+test "$(locale charmap 2>/dev/null | sed 's/gb/GB/')" = GB18030 ||
+ skip_ 'GB18030 charset support not detected'
+
+# U+10000 is encoded as 90 30 81 30. Its ASCII bytes must not be
+# counted again when the character is split across read buffers.
+padding=$(($IO_BUFSIZE - 1))
+head -c "$padding" /dev/zero | tr '\000' a > in || framework_failure_
+env printf '\x90\x30\x81\x30' >> in || framework_failure_
+printf '%s\n' "$IO_BUFSIZE" > exp || framework_failure_
+wc -m < in > out || fail=1
+compare exp out || fail=1
+
+Exit $fail
diff --git a/tests/wc/wc.pl b/tests/wc/wc.pl
index b57cd5145..a1b25b0b8 100755
--- a/tests/wc/wc.pl
+++ b/tests/wc/wc.pl
@@ -103,6 +103,17 @@ if (defined $single_byte_locale)
push @Tests, @new;
};
+# A split EM SPACE must not leave continuation bytes to start another word.
+# Separated here so we read from file and thus exercise IO_BUFSIZE.
+if (defined $mb_locale && $mb_locale ne 'none')
+ {
+ my $bufsize = getlimits ()->{IO_BUFSIZE};
+ push @Tests,
+ ['mb-split-space', '-w', '<',
+ {IN=>('a' x ($bufsize - 1)) . "\xe2\x80\x83"},
+ {OUT=>"1\n"}, {ENV=>"LC_ALL=$mb_locale"}];
+ }
+
my $save_temps = $ENV{DEBUG};
my $verbose = $ENV{VERBOSE};
--
2.55.0