* bootstrap.conf: Reference the new "utf8" module. * cfg.mk: Adjust checks as per mbbuf. * gl/lib/utf8.c: Include the implementation. * gl/lib/utf8.h: Add is_utf8_charset() and u8_count(). Note we don't give is_utf8_charset external-linkage due to its static cache. * gl/local.mk: Reference the new module files. * gl/modules/utf8: Add the module definition and dependencies. * src/cut.c: Use is_utf8_charset() and u8_count(). * src/numfmt.c: Use is_utf8_charset(). * src/system.h: Remove is_utf8_charset() definition. --- bootstrap.conf | 2 +- cfg.mk | 4 +-- gl/lib/utf8.c | 20 ++++++++++++ gl/lib/utf8.h | 81 +++++++++++++++++++++++++++++++++++++++++++++++++ gl/local.mk | 3 ++ gl/modules/utf8 | 30 ++++++++++++++++++ src/cut.c | 21 ++----------- src/numfmt.c | 1 + src/system.h | 14 --------- 9 files changed, 140 insertions(+), 36 deletions(-) create mode 100644 gl/lib/utf8.c create mode 100644 gl/lib/utf8.h create mode 100644 gl/modules/utf8
diff --git a/bootstrap.conf b/bootstrap.conf index d2bea314e..fb01aef9e 100644 --- a/bootstrap.conf +++ b/bootstrap.conf @@ -310,7 +310,6 @@ gnulib_modules=" uname unicodeio unistd-safer - unistr/u8-check unlink-busy unlinkat unlinkdir @@ -319,6 +318,7 @@ gnulib_modules=" update-copyright useless-if-before-free userspec + utf8 utimecmp utimens utimensat diff --git a/cfg.mk b/cfg.mk index 1c7be2463..8fe20500d 100644 --- a/cfg.mk +++ b/cfg.mk @@ -980,7 +980,7 @@ exclude_file_name_regexp--sc_prohibit_tab_based_indentation = \ $(tbi_1)|$(tbi_2)|$(tbi_3) exclude_file_name_regexp--sc_preprocessor_indentation = \ - ^(gl/lib/(rand-isaac|mbbuf)\.[ch]|gl/tests/test-rand-isaac\.c)$$|$(_ll) + ^(gl/lib/(rand-isaac|mbbuf|utf8)\.[ch]|gl/tests/test-rand-isaac\.c)$$|$(_ll) exclude_file_name_regexp--sc_prohibit_stat_st_blocks = \ ^(src/system\.h|tests/du/2g\.sh)$$ @@ -1042,4 +1042,4 @@ csiwl_2 = kno,ois,afile,whats,hda,indx,ot,nam,ist codespell_ignore_words_list = $(csiwl_1),$(csiwl_2) exclude_file_name_regexp--sc_codespell = \ ^(THANKS\.in|tests/pr/.*(F|tn?|l(o|m|i)|bl))$$ -exclude_file_name_regexp--sc_GPL_version = ^(gl/lib/mbbuf\.[hc])$$ +exclude_file_name_regexp--sc_GPL_version = ^(gl/lib/(mbbuf|utf8)\.[hc])$$ diff --git a/gl/lib/utf8.c b/gl/lib/utf8.c new file mode 100644 index 000000000..ad24739bc --- /dev/null +++ b/gl/lib/utf8.c @@ -0,0 +1,20 @@ +/* UTF-8 utilities. + Copyright (C) 2026 Free Software Foundation, Inc. + + This file is free software: you can redistribute it and/or modify + it under the terms of the GNU Lesser General Public License as + published by the Free Software Foundation; either version 2.1 of the + License, or (at your option) any later version. + + This file is distributed in the hope that it will be useful, + but WITHOUT ANY WARRANTY; without even the implied warranty of + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the + GNU Lesser General Public License for more details. + + You should have received a copy of the GNU Lesser General Public License + along with this program. If not, see <https://www.gnu.org/licenses/>. */ + +#include <config.h> + +#define UTF8_INLINE _GL_EXTERN_INLINE +#include "utf8.h" diff --git a/gl/lib/utf8.h b/gl/lib/utf8.h new file mode 100644 index 000000000..970d1b70f --- /dev/null +++ b/gl/lib/utf8.h @@ -0,0 +1,81 @@ +/* UTF-8 utilities. + Copyright (C) 2026 Free Software Foundation, Inc. + + This file is free software: you can redistribute it and/or modify + it under the terms of the GNU Lesser General Public License as + published by the Free Software Foundation; either version 2.1 of the + License, or (at your option) any later version. + + This file is distributed in the hope that it will be useful, + but WITHOUT ANY WARRANTY; without even the implied warranty of + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the + GNU Lesser General Public License for more details. + + You should have received a copy of the GNU Lesser General Public License + along with this program. If not, see <https://www.gnu.org/licenses/>. */ + +#ifndef _GL_UTF8_H +#define _GL_UTF8_H 1 + +#ifndef _GL_INLINE_HEADER_BEGIN +# error "Please include config.h first." +#endif + +#include <uchar.h> +#include <wchar.h> +#include <stdint.h> + +#include "idx.h" +#include "unistr.h" + +/* Return true if the current charset is UTF-8. */ +static inline bool +is_utf8_charset (void) +{ + static int is_utf8 = -1; + if (is_utf8 == -1) + { + char32_t w; + mbstate_t mbs; mbszero (&mbs); + is_utf8 = mbrtoc32 (&w, "\xe2\x9f\xb8", 3, &mbs) == 3 && w == 0x27F8; + } + return is_utf8; +} + +_GL_INLINE_HEADER_BEGIN +#ifndef UTF8_INLINE +# define UTF8_INLINE _GL_INLINE +#endif + +/* Count characters in the valid UTF-8 prefix of BUF's *NBYTES bytes. + Set *NBYTES to the length of that prefix, stopping before an invalid + or incomplete sequence. ASCII bytes, including NUL, count as characters. */ + +UTF8_INLINE idx_t +u8_count (char const *buf, idx_t *nbytes) +{ + uint8_t const *p = (uint8_t const *) buf; + idx_t n = *nbytes; + + /* Detect non-ASCII. */ + unsigned char bits = n ? p[0] : 0; + if (!(bits & 0x80)) + for (idx_t i = 0; i < n; i++) + bits |= p[i]; + if (!(bits & 0x80)) + return n; + + uint8_t const *invalid = u8_check (p, n); + if (invalid) + *nbytes = n = invalid - p; + + /* Count each non-continuation byte, i.e., each UTF-8 character. */ + idx_t count = 0; + for (idx_t i = 0; i < n; i++) + count += (p[i] & 0xc0) != 0x80; + return count; +} + +_GL_INLINE_HEADER_END + +#endif diff --git a/gl/local.mk b/gl/local.mk index 8befebe0a..d08efb263 100644 --- a/gl/local.mk +++ b/gl/local.mk @@ -51,6 +51,8 @@ gl/lib/strnumcmp.c \ gl/lib/strnumcmp.h \ gl/lib/targetdir.c \ gl/lib/targetdir.h \ +gl/lib/utf8.c \ +gl/lib/utf8.h \ gl/lib/xdectoimax.c \ gl/lib/xdectoint.c \ gl/lib/xdectoint.h \ @@ -78,6 +80,7 @@ gl/modules/skipchars \ gl/modules/smack \ gl/modules/strnumcmp \ gl/modules/targetdir \ +gl/modules/utf8 \ gl/modules/xdectoint \ gl/modules/xfts \ gl/tests/test-fadvise.c \ diff --git a/gl/modules/utf8 b/gl/modules/utf8 new file mode 100644 index 000000000..534b284e2 --- /dev/null +++ b/gl/modules/utf8 @@ -0,0 +1,30 @@ +Description: +UTF-8 helpers. + +Files: +lib/utf8.c +lib/utf8.h + +Depends-on: +bool +c99 +extern-inline +idx +mbrtoc32 +mbszero +unistr/u8-check +stdint-h + +configure.ac: + +Makefile.am: +lib_SOURCES += utf8.c utf8.h + +Include: +"utf8.h" + +License: +LGPLv2+ + +Maintainer: +all diff --git a/src/cut.c b/src/cut.c index 4d2bf9390..8e7ed3fac 100644 --- a/src/cut.c +++ b/src/cut.c @@ -38,7 +38,7 @@ #include "memchr2.h" #include "set-fields.h" -#include "unistr.h" +#include "utf8.h" /* The official name of this program (e.g., no 'g' prefix). */ #define PROGRAM_NAME "cut" @@ -951,24 +951,7 @@ cut_characters_mode (FILE *stream, bool byte_mode) continue; } - /* Detect non ASCII. */ - unsigned char bits = n ? p[0] : 0; - if (!(bits & 0x80)) - for (idx_t i = 0; i < n; i++) - bits |= p[i]; - uintmax_t count = n; - if (bits & 0x80) /* any UTF8 */ - { - uint8_t const *invalid = u8_check ((uint8_t const *) p, n); - if (invalid) - n = (char const *) invalid - p; - /* Count each non-continuation byte, i.e., each utf8 char. */ - count = 0; - for (idx_t i = 0; i < n; i++) - count += (to_uchar (p[i]) & 0xc0) != 0x80; - } - - idx += count; + idx += u8_count (p, &n); mbbuf_advance (&mbbuf, n); } diff --git a/src/numfmt.c b/src/numfmt.c index cdb1af472..692f20453 100644 --- a/src/numfmt.c +++ b/src/numfmt.c @@ -29,6 +29,7 @@ #include "quote.h" #include "skipchars.h" #include "system.h" +#include "utf8.h" #include "xstrtol.h" #include "set-fields.h" diff --git a/src/system.h b/src/system.h index 017491d98..66d0ab1b8 100644 --- a/src/system.h +++ b/src/system.h @@ -189,20 +189,6 @@ c32issep (char32_t wc) #endif } -/* Return true if the current charset is UTF-8. */ -static inline bool -is_utf8_charset (void) -{ - static int is_utf8 = -1; - if (is_utf8 == -1) - { - char32_t w; - mbstate_t mbs; mbszero (&mbs); - is_utf8 = mbrtoc32 (&w, "\xe2\x9f\xb8", 3, &mbs) == 3 && w == 0x27F8; - } - return is_utf8; -} - #include <locale.h> /* Take care of NLS matters. */ -- 2.55.0
