* bootstrap.conf: Reference the new "utf8" module.
* cfg.mk: Adjust checks as per mbbuf.
* gl/lib/utf8.c: Include the implementation.
* gl/lib/utf8.h: Add is_utf8_charset() and u8_count().
Note we don't give is_utf8_charset external-linkage
due to its static cache.
* gl/local.mk: Reference the new module files.
* gl/modules/utf8: Add the module definition and dependencies.
* src/cut.c: Use is_utf8_charset() and u8_count().
* src/numfmt.c: Use is_utf8_charset().
* src/system.h: Remove is_utf8_charset() definition.
---
 bootstrap.conf  |  2 +-
 cfg.mk          |  4 +--
 gl/lib/utf8.c   | 20 ++++++++++++
 gl/lib/utf8.h   | 81 +++++++++++++++++++++++++++++++++++++++++++++++++
 gl/local.mk     |  3 ++
 gl/modules/utf8 | 30 ++++++++++++++++++
 src/cut.c       | 21 ++-----------
 src/numfmt.c    |  1 +
 src/system.h    | 14 ---------
 9 files changed, 140 insertions(+), 36 deletions(-)
 create mode 100644 gl/lib/utf8.c
 create mode 100644 gl/lib/utf8.h
 create mode 100644 gl/modules/utf8

diff --git a/bootstrap.conf b/bootstrap.conf
index d2bea314e..fb01aef9e 100644
--- a/bootstrap.conf
+++ b/bootstrap.conf
@@ -310,7 +310,6 @@ gnulib_modules="
   uname
   unicodeio
   unistd-safer
-  unistr/u8-check
   unlink-busy
   unlinkat
   unlinkdir
@@ -319,6 +318,7 @@ gnulib_modules="
   update-copyright
   useless-if-before-free
   userspec
+  utf8
   utimecmp
   utimens
   utimensat
diff --git a/cfg.mk b/cfg.mk
index 1c7be2463..8fe20500d 100644
--- a/cfg.mk
+++ b/cfg.mk
@@ -980,7 +980,7 @@ exclude_file_name_regexp--sc_prohibit_tab_based_indentation 
= \
   $(tbi_1)|$(tbi_2)|$(tbi_3)
 
 exclude_file_name_regexp--sc_preprocessor_indentation = \
-  ^(gl/lib/(rand-isaac|mbbuf)\.[ch]|gl/tests/test-rand-isaac\.c)$$|$(_ll)
+  ^(gl/lib/(rand-isaac|mbbuf|utf8)\.[ch]|gl/tests/test-rand-isaac\.c)$$|$(_ll)
 exclude_file_name_regexp--sc_prohibit_stat_st_blocks = \
   ^(src/system\.h|tests/du/2g\.sh)$$
 
@@ -1042,4 +1042,4 @@ csiwl_2 = kno,ois,afile,whats,hda,indx,ot,nam,ist
 codespell_ignore_words_list = $(csiwl_1),$(csiwl_2)
 exclude_file_name_regexp--sc_codespell = \
   ^(THANKS\.in|tests/pr/.*(F|tn?|l(o|m|i)|bl))$$
-exclude_file_name_regexp--sc_GPL_version = ^(gl/lib/mbbuf\.[hc])$$
+exclude_file_name_regexp--sc_GPL_version = ^(gl/lib/(mbbuf|utf8)\.[hc])$$
diff --git a/gl/lib/utf8.c b/gl/lib/utf8.c
new file mode 100644
index 000000000..ad24739bc
--- /dev/null
+++ b/gl/lib/utf8.c
@@ -0,0 +1,20 @@
+/* UTF-8 utilities.
+   Copyright (C) 2026 Free Software Foundation, Inc.
+
+   This file is free software: you can redistribute it and/or modify
+   it under the terms of the GNU Lesser General Public License as
+   published by the Free Software Foundation; either version 2.1 of the
+   License, or (at your option) any later version.
+
+   This file is distributed in the hope that it will be useful,
+   but WITHOUT ANY WARRANTY; without even the implied warranty of
+   MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
+   GNU Lesser General Public License for more details.
+
+   You should have received a copy of the GNU Lesser General Public License
+   along with this program.  If not, see <https://www.gnu.org/licenses/>.  */
+
+#include <config.h>
+
+#define UTF8_INLINE _GL_EXTERN_INLINE
+#include "utf8.h"
diff --git a/gl/lib/utf8.h b/gl/lib/utf8.h
new file mode 100644
index 000000000..970d1b70f
--- /dev/null
+++ b/gl/lib/utf8.h
@@ -0,0 +1,81 @@
+/* UTF-8 utilities.
+   Copyright (C) 2026 Free Software Foundation, Inc.
+
+   This file is free software: you can redistribute it and/or modify
+   it under the terms of the GNU Lesser General Public License as
+   published by the Free Software Foundation; either version 2.1 of the
+   License, or (at your option) any later version.
+
+   This file is distributed in the hope that it will be useful,
+   but WITHOUT ANY WARRANTY; without even the implied warranty of
+   MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
+   GNU Lesser General Public License for more details.
+
+   You should have received a copy of the GNU Lesser General Public License
+   along with this program.  If not, see <https://www.gnu.org/licenses/>.  */
+
+#ifndef _GL_UTF8_H
+#define _GL_UTF8_H 1
+
+#ifndef _GL_INLINE_HEADER_BEGIN
+# error "Please include config.h first."
+#endif
+
+#include <uchar.h>
+#include <wchar.h>
+#include <stdint.h>
+
+#include "idx.h"
+#include "unistr.h"
+
+/* Return true if the current charset is UTF-8.  */
+static inline bool
+is_utf8_charset (void)
+{
+  static int is_utf8 = -1;
+  if (is_utf8 == -1)
+    {
+      char32_t w;
+      mbstate_t mbs; mbszero (&mbs);
+      is_utf8 = mbrtoc32 (&w, "\xe2\x9f\xb8", 3, &mbs) == 3 && w == 0x27F8;
+    }
+  return is_utf8;
+}
+
+_GL_INLINE_HEADER_BEGIN
+#ifndef UTF8_INLINE
+# define UTF8_INLINE _GL_INLINE
+#endif
+
+/* Count characters in the valid UTF-8 prefix of BUF's *NBYTES bytes.
+   Set *NBYTES to the length of that prefix, stopping before an invalid
+   or incomplete sequence.  ASCII bytes, including NUL, count as characters.  
*/
+
+UTF8_INLINE idx_t
+u8_count (char const *buf, idx_t *nbytes)
+{
+  uint8_t const *p = (uint8_t const *) buf;
+  idx_t n = *nbytes;
+
+  /* Detect non-ASCII.  */
+  unsigned char bits = n ? p[0] : 0;
+  if (!(bits & 0x80))
+    for (idx_t i = 0; i < n; i++)
+      bits |= p[i];
+  if (!(bits & 0x80))
+    return n;
+
+  uint8_t const *invalid = u8_check (p, n);
+  if (invalid)
+    *nbytes = n = invalid - p;
+
+  /* Count each non-continuation byte, i.e., each UTF-8 character.  */
+  idx_t count = 0;
+  for (idx_t i = 0; i < n; i++)
+    count += (p[i] & 0xc0) != 0x80;
+  return count;
+}
+
+_GL_INLINE_HEADER_END
+
+#endif
diff --git a/gl/local.mk b/gl/local.mk
index 8befebe0a..d08efb263 100644
--- a/gl/local.mk
+++ b/gl/local.mk
@@ -51,6 +51,8 @@ gl/lib/strnumcmp.c \
 gl/lib/strnumcmp.h \
 gl/lib/targetdir.c \
 gl/lib/targetdir.h \
+gl/lib/utf8.c \
+gl/lib/utf8.h \
 gl/lib/xdectoimax.c \
 gl/lib/xdectoint.c \
 gl/lib/xdectoint.h \
@@ -78,6 +80,7 @@ gl/modules/skipchars \
 gl/modules/smack \
 gl/modules/strnumcmp \
 gl/modules/targetdir \
+gl/modules/utf8 \
 gl/modules/xdectoint \
 gl/modules/xfts \
 gl/tests/test-fadvise.c \
diff --git a/gl/modules/utf8 b/gl/modules/utf8
new file mode 100644
index 000000000..534b284e2
--- /dev/null
+++ b/gl/modules/utf8
@@ -0,0 +1,30 @@
+Description:
+UTF-8 helpers.
+
+Files:
+lib/utf8.c
+lib/utf8.h
+
+Depends-on:
+bool
+c99
+extern-inline
+idx
+mbrtoc32
+mbszero
+unistr/u8-check
+stdint-h
+
+configure.ac:
+
+Makefile.am:
+lib_SOURCES += utf8.c utf8.h
+
+Include:
+"utf8.h"
+
+License:
+LGPLv2+
+
+Maintainer:
+all
diff --git a/src/cut.c b/src/cut.c
index 4d2bf9390..8e7ed3fac 100644
--- a/src/cut.c
+++ b/src/cut.c
@@ -38,7 +38,7 @@
 #include "memchr2.h"
 
 #include "set-fields.h"
-#include "unistr.h"
+#include "utf8.h"
 
 /* The official name of this program (e.g., no 'g' prefix).  */
 #define PROGRAM_NAME "cut"
@@ -951,24 +951,7 @@ cut_characters_mode (FILE *stream, bool byte_mode)
               continue;
             }
 
-          /* Detect non ASCII.  */
-          unsigned char bits = n ? p[0] : 0;
-          if (!(bits & 0x80))
-            for (idx_t i = 0; i < n; i++)
-              bits |= p[i];
-          uintmax_t count = n;
-          if (bits & 0x80) /* any UTF8 */
-            {
-              uint8_t const *invalid = u8_check ((uint8_t const *) p, n);
-              if (invalid)
-                n = (char const *) invalid - p;
-              /* Count each non-continuation byte, i.e., each utf8 char.  */
-              count = 0;
-              for (idx_t i = 0; i < n; i++)
-                count += (to_uchar (p[i]) & 0xc0) != 0x80;
-            }
-
-          idx += count;
+          idx += u8_count (p, &n);
           mbbuf_advance (&mbbuf, n);
         }
 
diff --git a/src/numfmt.c b/src/numfmt.c
index cdb1af472..692f20453 100644
--- a/src/numfmt.c
+++ b/src/numfmt.c
@@ -29,6 +29,7 @@
 #include "quote.h"
 #include "skipchars.h"
 #include "system.h"
+#include "utf8.h"
 #include "xstrtol.h"
 
 #include "set-fields.h"
diff --git a/src/system.h b/src/system.h
index 017491d98..66d0ab1b8 100644
--- a/src/system.h
+++ b/src/system.h
@@ -189,20 +189,6 @@ c32issep (char32_t wc)
 #endif
 }
 
-/* Return true if the current charset is UTF-8.  */
-static inline bool
-is_utf8_charset (void)
-{
-  static int is_utf8 = -1;
-  if (is_utf8 == -1)
-    {
-      char32_t w;
-      mbstate_t mbs; mbszero (&mbs);
-      is_utf8 = mbrtoc32 (&w, "\xe2\x9f\xb8", 3, &mbs) == 3 && w == 0x27F8;
-    }
-  return is_utf8;
-}
-
 #include <locale.h>
 
 /* Take care of NLS matters.  */
-- 
2.55.0


Reply via email to