Script 'mail_helper' called by obssrc Hello community, here is the log from the commit of package simdutf for openSUSE:Factory checked in at 2026-09-23 14:32:38 ++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++ Comparing /work/SRC/openSUSE:Factory/simdutf (Old) and /work/SRC/openSUSE:Factory/.simdutf.new.383539 (New) ++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Package is "simdutf" Wed Sep 23 14:32:38 2026 rev:7 rq:1379636 version:9.2.0 Changes: -------- --- /work/SRC/openSUSE:Factory/simdutf/simdutf.changes 2026-09-15 17:09:54.623079767 +0200 +++ /work/SRC/openSUSE:Factory/.simdutf.new.383539/simdutf.changes 2026-09-23 14:33:20.730310982 +0200 @@ -1,0 +2,10 @@ +Wed Sep 16 18:50:59 UTC 2026 - Dominique Leuenberger <[email protected]> + +- Update to version 9.2.0: + + arm64: test comparison masks with vmaxvq_u32 instead of + shrn+fcmp + + add detailed results to safe UTF conversions +- Update lib_ver to 36.0.0 and so_ver to 36: follow upstream + changes. + +------------------------------------------------------------------- Old: ---- simdutf-9.1.2.tar.xz New: ---- simdutf-9.2.0.tar.xz ++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++ Other differences: ------------------ ++++++ simdutf.spec ++++++ --- /var/tmp/diff_new_pack.uSKp4l/_old 2026-09-23 14:33:22.302376702 +0200 +++ /var/tmp/diff_new_pack.uSKp4l/_new 2026-09-23 14:33:22.304376786 +0200 @@ -16,10 +16,10 @@ # -%define lib_ver 35.0.0 -%define so_ver 35 +%define lib_ver 36.0.0 +%define so_ver 36 Name: simdutf -Version: 9.1.2 +Version: 9.2.0 Release: 0 Summary: Unicode validation and transcoding at billions of characters per second ++++++ _scmsync.obsinfo ++++++ --- /var/tmp/diff_new_pack.uSKp4l/_old 2026-09-23 14:33:22.340378291 +0200 +++ /var/tmp/diff_new_pack.uSKp4l/_new 2026-09-23 14:33:22.343378416 +0200 @@ -1,7 +1,7 @@ -mtime: 1789124538 -commit: 476758be511f1e54b9c073a85eeb05f3a97682883b08d05b825bec595632d160 +mtime: 1789585096 +commit: 6a05c2e90060df022476ec1ba90b9d4c848ab1d38953ed2517a7ba4bb22d6fc1 url: https://src.opensuse.org/GNOME/simdutf -revision: 476758be511f1e54b9c073a85eeb05f3a97682883b08d05b825bec595632d160 +revision: 6a05c2e90060df022476ec1ba90b9d4c848ab1d38953ed2517a7ba4bb22d6fc1 trackingbranch: factory projectscmsync: https://src.opensuse.org/GNOME/_ObsPrj ++++++ _service ++++++ --- /var/tmp/diff_new_pack.uSKp4l/_old 2026-09-23 14:33:22.363379252 +0200 +++ /var/tmp/diff_new_pack.uSKp4l/_new 2026-09-23 14:33:22.366379378 +0200 @@ -3,7 +3,7 @@ <service name="obs_scm" mode="manual"> <param name="scm">git</param> <param name="url">https://github.com/simdutf/simdutf.git</param> - <param name="revision">v9.1.2</param> + <param name="revision">v9.2.0</param> <param name="versionformat">@PARENT_TAG@+@TAG_OFFSET@</param> <param name="versionrewrite-pattern">v?(.*)\+0</param> <param name="versionrewrite-replacement">\1</param> ++++++ build.specials.obscpio ++++++ ++++++ build.specials.obscpio ++++++ diff -urN '--exclude=CVS' '--exclude=.cvsignore' '--exclude=.svn' '--exclude=.svnignore' old/.gitignore new/.gitignore --- old/.gitignore 1970-01-01 01:00:00.000000000 +0100 +++ new/.gitignore 2026-09-16 20:58:16.000000000 +0200 @@ -0,0 +1,5 @@ +*.obscpio +*.osc +_build.* +.pbuild +osc-collab.* ++++++ simdutf-9.1.2.tar.xz -> simdutf-9.2.0.tar.xz ++++++ diff -urN '--exclude=CVS' '--exclude=.cvsignore' '--exclude=.svn' '--exclude=.svnignore' old/simdutf-9.1.2/.github/workflows/release.yml new/simdutf-9.2.0/.github/workflows/release.yml --- old/simdutf-9.1.2/.github/workflows/release.yml 1970-01-01 01:00:00.000000000 +0100 +++ new/simdutf-9.2.0/.github/workflows/release.yml 2026-09-15 17:49:26.000000000 +0200 @@ -0,0 +1,136 @@ +name: release + +# Manually triggered release: bumps the version (patch, minor or major), +# regenerates the amalgamated single-header files, commits and tags on +# master, then publishes a GitHub release with the single-header files +# attached (simdutf.h, simdutf.cpp, singleheader.zip). + +on: + workflow_dispatch: + inputs: + release_type: + description: "Which part of the version to bump" + required: true + type: choice + default: patch + options: + - patch + - minor + - major + dry_run: + description: "Do everything except pushing the commit/tag and creating the release" + required: false + type: boolean + default: false + +permissions: + contents: write + actions: write + +concurrency: + group: release + cancel-in-progress: false + +jobs: + release: + runs-on: ubuntu-24.04 + steps: + - uses: actions/checkout@v6 + with: + ref: master + fetch-depth: 0 # release.py needs the tags to find the last version + + - name: Compute the next version + id: version + env: + RELEASE_TYPE: ${{ inputs.release_type }} + run: | + set -euo pipefail + last=$(git describe --tags --abbrev=0 --match 'v[0-9]*') + echo "last tag: $last" + IFS=. read -r major minor patch <<< "${last#v}" + case "$RELEASE_TYPE" in + major) major=$((major + 1)); minor=0; patch=0 ;; + minor) minor=$((minor + 1)); patch=0 ;; + patch) patch=$((patch + 1)) ;; + *) echo "unknown release type: $RELEASE_TYPE" >&2; exit 1 ;; + esac + version="$major.$minor.$patch" + echo "next version: $version" + if git rev-parse -q --verify "refs/tags/v$version" >/dev/null; then + echo "tag v$version already exists" >&2 + exit 1 + fi + echo "version=$version" >> "$GITHUB_OUTPUT" + echo "previous=${last}" >> "$GITHUB_OUTPUT" + + - name: Bump the version and regenerate the single-header files + run: | + set -euo pipefail + python3 scripts/release.py "${{ steps.version.outputs.version }}" + rm -f CMakeLists.txt.bak Doxyfile.bak README.md.bak + git --no-pager diff --stat + ls -l singleheader/simdutf.h singleheader/simdutf.cpp singleheader/singleheader.zip + + - name: Check that the single-header files compile and run + run: | + set -euo pipefail + cd singleheader + c++ -o amalgamation_demo amalgamation_demo.cpp -std=c++17 + ./amalgamation_demo + c++ -c simdutf.cpp -std=c++17 + cc -c amalgamation_demo.c + c++ amalgamation_demo.o simdutf.o -o cdemo + ./cdemo + + - name: Commit and tag + env: + VERSION: ${{ steps.version.outputs.version }} + run: | + set -euo pipefail + git config user.name "github-actions[bot]" + git config user.email "41898282+github-actions[bot]@users.noreply.github.com" + git add CMakeLists.txt Doxyfile README.md include/simdutf/simdutf_version.h + git commit -m "Release v$VERSION" + git tag -a "v$VERSION" -m "version $VERSION" + git --no-pager show --stat HEAD + + - name: Push the commit and tag + if: ${{ !inputs.dry_run }} + run: | + set -euo pipefail + git push origin master + git push origin "v${{ steps.version.outputs.version }}" + + - name: Create the GitHub release + if: ${{ !inputs.dry_run }} + env: + GH_TOKEN: ${{ github.token }} + VERSION: ${{ steps.version.outputs.version }} + PREVIOUS: ${{ steps.version.outputs.previous }} + run: | + set -euo pipefail + gh release create "v$VERSION" \ + --title "Version $VERSION" \ + --generate-notes \ + --notes-start-tag "$PREVIOUS" \ + singleheader/simdutf.h \ + singleheader/simdutf.cpp \ + singleheader/singleheader.zip + + # Releases created with GITHUB_TOKEN do not fire the `release`/`push` events + # that documentation.yml listens to, so dispatch it explicitly. + - name: Publish the API documentation + if: ${{ !inputs.dry_run }} + env: + GH_TOKEN: ${{ github.token }} + run: gh workflow run documentation.yml --ref "v${{ steps.version.outputs.version }}" + + - name: Upload the release files as a workflow artifact + uses: actions/upload-artifact@v4 + with: + name: singleheader-v${{ steps.version.outputs.version }} + path: | + singleheader/simdutf.h + singleheader/simdutf.cpp + singleheader/singleheader.zip diff -urN '--exclude=CVS' '--exclude=.cvsignore' '--exclude=.svn' '--exclude=.svnignore' old/simdutf-9.1.2/CMakeLists.txt new/simdutf-9.2.0/CMakeLists.txt --- old/simdutf-9.1.2/CMakeLists.txt 2026-09-11 04:41:45.000000000 +0200 +++ new/simdutf-9.2.0/CMakeLists.txt 2026-09-15 17:49:26.000000000 +0200 @@ -3,7 +3,7 @@ project(simdutf DESCRIPTION "Fast Unicode validation, transcoding and processing" LANGUAGES C CXX - VERSION 9.1.2 + VERSION 9.2.0 ) include (TestBigEndian) @@ -23,8 +23,8 @@ include(CTest) include(cmake/simdutf-flags.cmake) -set(SIMDUTF_LIB_VERSION "35.0.0" CACHE STRING "simdutf library version") -set(SIMDUTF_LIB_SOVERSION "35" CACHE STRING "simdutf library soversion") +set(SIMDUTF_LIB_VERSION "36.0.0" CACHE STRING "simdutf library version") +set(SIMDUTF_LIB_SOVERSION "36" CACHE STRING "simdutf library soversion") option(SIMDUTF_TESTS "Whether the tests are included as part of the CMake Build." ON) option(SIMDUTF_FAST_TESTS "Whether parameters to the tests are adapted to make them execute quicker" OFF) option(SIMDUTF_TOOL_TESTS "Whether to tests the tools" OFF) diff -urN '--exclude=CVS' '--exclude=.cvsignore' '--exclude=.svn' '--exclude=.svnignore' old/simdutf-9.1.2/Doxyfile new/simdutf-9.2.0/Doxyfile --- old/simdutf-9.1.2/Doxyfile 2026-09-11 04:41:45.000000000 +0200 +++ new/simdutf-9.2.0/Doxyfile 2026-09-15 17:49:26.000000000 +0200 @@ -38,7 +38,7 @@ # could be handy for archiving the generated documentation or if some version # control system is used. -PROJECT_NUMBER = "9.1.2" +PROJECT_NUMBER = "9.2.0" # Using the PROJECT_BRIEF tag one can provide an optional one line description # for a project that appears at the top of each page and should give viewer a diff -urN '--exclude=CVS' '--exclude=.cvsignore' '--exclude=.svn' '--exclude=.svnignore' old/simdutf-9.1.2/README.md new/simdutf-9.2.0/README.md --- old/simdutf-9.1.2/README.md 2026-09-11 04:41:45.000000000 +0200 +++ new/simdutf-9.2.0/README.md 2026-09-15 17:49:26.000000000 +0200 @@ -149,7 +149,7 @@ 1. Pull the library in a directory ``` - wget https://github.com/simdutf/simdutf/releases/download/v9.1.2/singleheader.zip + wget https://github.com/simdutf/simdutf/releases/download/v9.2.0/singleheader.zip unzip singleheader.zip ``` You can replace `wget` by `curl -OL https://...` if you prefer. @@ -190,7 +190,7 @@ ## Single-header version -You can create a single-header version of the library where all of the code is put into two files (`simdutf.h` and `simdutf.cpp`). We publish a zip archive containing these files, e.g., see https://github.com/simdutf/simdutf/releases/download/v9.1.2/singleheader.zip +You can create a single-header version of the library where all of the code is put into two files (`simdutf.h` and `simdutf.cpp`). We publish a zip archive containing these files, e.g., see https://github.com/simdutf/simdutf/releases/download/v9.2.0/singleheader.zip You may generate it on your own using a Python script. @@ -1104,6 +1104,23 @@ simdutf_warn_unused size_t convert_latin1_to_utf8_safe(const char * input, size_t length, char* utf8_output, size_t utf8_len) noexcept; /** + * Convert a Latin1 string into a size-limited UTF-8 buffer and report how much + * input was consumed and output was written. + * + * We write as many complete characters as possible. The returned error is + * SUCCESS if all input was consumed, or OUTPUT_BUFFER_TOO_SMALL otherwise. + * + * @param input the Latin1 string to convert + * @param length the length of the string in bytes + * @param utf8_output the pointer to the output buffer + * @param utf8_len the maximum output length + * @return a full_result with error, input_count and output_count + */ +simdutf_warn_unused full_result convert_latin1_to_utf8_safe_with_details( + const char *input, size_t length, char *utf8_output, + size_t utf8_len) noexcept; + +/** * Using native endianness, convert a Latin1 string into a UTF-16 string. * * @param input the Latin1 string to convert @@ -1264,6 +1281,25 @@ size_t utf8_len) noexcept; /** + * Convert a possibly broken UTF-16 string into a size-limited UTF-8 buffer and + * report how much input was consumed and output was written. + * + * We write as many complete characters as possible while validating the input. + * The returned error is SUCCESS if all input was consumed, + * OUTPUT_BUFFER_TOO_SMALL if the next character does not fit, or SURROGATE if + * an unpaired surrogate was found. + * + * @param input the UTF-16 string to convert + * @param length the length in 16-bit code units + * @param utf8_output the pointer to the output buffer + * @param utf8_len the maximum output length + * @return a full_result with error, input_count and output_count + */ +simdutf_warn_unused full_result convert_utf16_to_utf8_safe_with_details( + const char16_t *input, size_t length, char *utf8_output, + size_t utf8_len) noexcept; + +/** * Using native endianness, convert possibly broken UTF-16 string into UTF-8 * string, replacing unpaired surrogates with the Unicode replacement character * U+FFFD. @@ -1282,6 +1318,26 @@ const char16_t *input, size_t length, char *utf8_buffer) noexcept; /** + * Convert a possibly broken UTF-16 string into a size-limited UTF-8 buffer, + * replacing unpaired surrogates with U+FFFD and reporting how much input was + * consumed and output was written. + * + * We write as many complete characters as possible. The returned error is + * SUCCESS if all input was consumed, or OUTPUT_BUFFER_TOO_SMALL if the next + * character or replacement does not fit. + * + * @param input the UTF-16 string to convert + * @param length the length in 16-bit code units + * @param utf8_output the pointer to the output buffer + * @param utf8_len the maximum output length + * @return a full_result with error, input_count and output_count + */ +simdutf_warn_unused full_result +convert_utf16_to_utf8_with_replacement_safe(const char16_t *input, + size_t length, char *utf8_output, + size_t utf8_len) noexcept; + +/** * Using native endianness, convert possibly broken UTF-16 string into Latin1 string. * If the string cannot be represented as Latin1, an error * is returned. @@ -1981,7 +2037,7 @@ ## Cost of the safe conversion functions -The `_safe` conversion variants (`convert_latin1_to_utf8_safe` and `convert_utf16_to_utf8_safe`) never write past the output capacity you give them. Because these functions cannot assume that there is enough output buffer space, they cannot proceed in the most efficient manner. For example, they may be forced to split the work into chunks. If the inputs span megabytes, this overhead is negligible. Unfortunately, for small inputs, it can be significant. For example, the `convert_utf16_to_utf8_safe` function is up to 3 times slower than `convert_utf16_to_utf8` on ASCII inputs of a few hundred code units in some tests. For optimal performance, you should allocate at least as much memory as the `utf8_length_from_latin1` or `utf8_length_from_utf16` functions indicate and directly call the `convert_latin1_to_utf8` and `convert_utf16_to_utf8` functions, especially if you expect to have short inputs. +The `_safe` conversion variants (`convert_latin1_to_utf8_safe` and `convert_utf16_to_utf8_safe`) never write past the output capacity you give them. The corresponding `_safe_with_details` variants additionally return the number of input code units consumed and output bytes written. Because these functions cannot assume that there is enough output buffer space, they cannot proceed in the most efficient manner. For example, they may be forced to split the work into chunks. If the inputs span megabytes, this overhead is negligible. Unfortunately, for small inputs, it can be significant. For example, the `convert_utf16_to_utf8_safe` function is up to 3 times slower than `convert_utf16_to_utf8` on ASCII inputs of a few hundred code units in some tests. For optimal performance, you should allocate at least as much memory as the `utf8_length_from_latin1` or `utf8_length_from_utf16` functions indicate and directly call the `convert_latin1_to_utf8` and `convert_utf16_to_utf8` functions, espe cially if you expect to have short inputs. The base64 decoding functions have their own safe variant, `base64_to_binary_safe`, which takes the output capacity as an in-out parameter. It does not need to split the work into chunks: it determines in a single step how much of the input fits in the output buffer, decodes that part with the fast function, and leaves only the remainder to a scalar decoder. Its overhead is therefore normally negligible, and we measure it to be as fast as `base64_to_binary` on clean base64 inputs at all sizes. The exception is base64 containing ASCII whitespace, because whitespace breaks the relationship between the input length and the output length: a short input of a few dozen characters with 5% whitespace can be nearly 3 times slower, although the difference largely disappears for inputs spanning a kilobyte or more. The `atomic_base64_to_binary_safe` function is more expensive: it decodes into a small temporary buffer and then copies the result to the output with relaxed atomic writes, so that o ther threads never observe partially written data. Every output byte is thus written twice, and this cost does not go away with larger inputs: we measure it to be 1.5 to 1.8 times slower than `base64_to_binary` on inputs of a kilobyte or more, including inputs spanning megabytes. You should only use it when the output buffer might be accessed concurrently. diff -urN '--exclude=CVS' '--exclude=.cvsignore' '--exclude=.svn' '--exclude=.svnignore' old/simdutf-9.1.2/include/simdutf/implementation.h new/simdutf-9.2.0/include/simdutf/implementation.h --- old/simdutf-9.1.2/include/simdutf/implementation.h 2026-09-11 04:41:45.000000000 +0200 +++ new/simdutf-9.2.0/include/simdutf/implementation.h 2026-09-15 17:49:26.000000000 +0200 @@ -897,6 +897,41 @@ } } #endif // SIMDUTF_SPAN + +/** + * Convert a Latin1 string into a size-limited UTF-8 buffer and report how much + * input was consumed and output was written. + * + * We write as many complete characters as possible. The returned error is + * SUCCESS if all input was consumed, or OUTPUT_BUFFER_TOO_SMALL otherwise. + * + * @param input the Latin1 string to convert + * @param length the length of the string in bytes + * @param utf8_output the pointer to the output buffer + * @param utf8_len the maximum output length + * @return a full_result with error, input_count and output_count + */ +simdutf_warn_unused full_result convert_latin1_to_utf8_safe_with_details( + const char *input, size_t length, char *utf8_output, + size_t utf8_len) noexcept; + #if SIMDUTF_SPAN +simdutf_really_inline simdutf_warn_unused simdutf_constexpr23 full_result +convert_latin1_to_utf8_safe_with_details( + const detail::input_span_of_byte_like auto &input, + detail::output_span_of_byte_like auto &&utf8_output) noexcept { + #if SIMDUTF_CPLUSPLUS23 + if consteval { + return scalar::latin1_to_utf8::convert_safe_with_details_constexpr( + input.data(), input.size(), utf8_output.data(), utf8_output.size()); + } else + #endif + { + return convert_latin1_to_utf8_safe_with_details( + reinterpret_cast<const char *>(input.data()), input.size(), + reinterpret_cast<char *>(utf8_output.data()), utf8_output.size()); + } +} + #endif // SIMDUTF_SPAN #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_LATIN1 #if SIMDUTF_FEATURE_UTF16 && SIMDUTF_FEATURE_LATIN1 @@ -1879,6 +1914,47 @@ } } #endif // SIMDUTF_SPAN + +/** + * Convert a possibly broken UTF-16 string into a size-limited UTF-8 buffer and + * report how much input was consumed and output was written. + * + * We write as many complete characters as possible while validating the input. + * The returned error is SUCCESS if all input was consumed, + * OUTPUT_BUFFER_TOO_SMALL if the next character does not fit, or SURROGATE if + * an unpaired surrogate was found. + * + * @param input the UTF-16 string to convert + * @param length the length in 16-bit code units + * @param utf8_output the pointer to the output buffer + * @param utf8_len the maximum output length + * @return a full_result with error, input_count and output_count + */ +simdutf_warn_unused full_result convert_utf16_to_utf8_safe_with_details( + const char16_t *input, size_t length, char *utf8_output, + size_t utf8_len) noexcept; + #if SIMDUTF_SPAN +simdutf_really_inline simdutf_warn_unused simdutf_constexpr23 full_result +convert_utf16_to_utf8_safe_with_details( + std::span<const char16_t> utf16_input, + detail::output_span_of_byte_like auto &&utf8_output) noexcept { + #if SIMDUTF_CPLUSPLUS23 + if consteval { + if (utf16_input.empty()) { + return full_result(error_code::SUCCESS, 0, 0); + } + return scalar::utf16_to_utf8::convert_with_errors<endianness::NATIVE, true>( + utf16_input.data(), utf16_input.size(), utf8_output.data(), + utf8_output.size()); + } else + #endif + { + return convert_utf16_to_utf8_safe_with_details( + utf16_input.data(), utf16_input.size(), + reinterpret_cast<char *>(utf8_output.data()), utf8_output.size()); + } +} + #endif // SIMDUTF_SPAN #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF16 #if SIMDUTF_FEATURE_UTF16 && SIMDUTF_FEATURE_LATIN1 @@ -2395,6 +2471,44 @@ } } #endif // SIMDUTF_SPAN + +/** + * Convert a possibly broken UTF-16 string into a size-limited UTF-8 buffer, + * replacing unpaired surrogates with U+FFFD and reporting how much input was + * consumed and output was written. + * + * We write as many complete characters as possible. The returned error is + * SUCCESS if all input was consumed, or OUTPUT_BUFFER_TOO_SMALL if the next + * character or replacement does not fit. + * + * @param input the UTF-16 string to convert + * @param length the length in 16-bit code units + * @param utf8_output the pointer to the output buffer + * @param utf8_len the maximum output length + * @return a full_result with error, input_count and output_count + */ +simdutf_warn_unused full_result convert_utf16_to_utf8_with_replacement_safe( + const char16_t *input, size_t length, char *utf8_output, + size_t utf8_len) noexcept; + #if SIMDUTF_SPAN +simdutf_really_inline simdutf_warn_unused simdutf_constexpr23 full_result +convert_utf16_to_utf8_with_replacement_safe( + std::span<const char16_t> utf16_input, + detail::output_span_of_byte_like auto &&utf8_output) noexcept { + #if SIMDUTF_CPLUSPLUS23 + if consteval { + return scalar::utf16_to_utf8::convert_with_replacement_safe< + endianness::NATIVE>(utf16_input.data(), utf16_input.size(), + utf8_output.data(), utf8_output.size()); + } else + #endif + { + return convert_utf16_to_utf8_with_replacement_safe( + utf16_input.data(), utf16_input.size(), + reinterpret_cast<char *>(utf8_output.data()), utf8_output.size()); + } +} + #endif // SIMDUTF_SPAN #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF16 #if SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF16 diff -urN '--exclude=CVS' '--exclude=.cvsignore' '--exclude=.svn' '--exclude=.svnignore' old/simdutf-9.1.2/include/simdutf/scalar/latin1_to_utf8/latin1_to_utf8.h new/simdutf-9.2.0/include/simdutf/scalar/latin1_to_utf8/latin1_to_utf8.h --- old/simdutf-9.1.2/include/simdutf/scalar/latin1_to_utf8/latin1_to_utf8.h 2026-09-11 04:41:45.000000000 +0200 +++ new/simdutf-9.2.0/include/simdutf/scalar/latin1_to_utf8/latin1_to_utf8.h 2026-09-15 17:49:26.000000000 +0200 @@ -116,6 +116,28 @@ return utf8_pos; } +inline full_result convert_safe_with_details(const char *buf, size_t len, + char *utf8_output, + size_t utf8_len) { + const size_t output_count = convert_safe(buf, len, utf8_output, utf8_len); + // Recover the consumed input count from the completed output. The runtime + // safe converter uses this helper only for its short scalar tail. + size_t input_count = 0; + size_t counted_output = 0; + while (input_count < len) { + const size_t width = + uint8_t(buf[input_count]) < uint8_t(0x80) ? size_t(1) : size_t(2); + if (counted_output + width > output_count) { + break; + } + input_count++; + counted_output += width; + } + return full_result(input_count == len ? error_code::SUCCESS + : error_code::OUTPUT_BUFFER_TOO_SMALL, + input_count, output_count); +} + template <typename InputPtr, typename OutputPtr> #if SIMDUTF_CPLUSPLUS20 requires(simdutf::detail::indexes_into_byte_like<InputPtr> && @@ -144,6 +166,31 @@ return utf8_pos; } +template <typename InputPtr, typename OutputPtr> +#if SIMDUTF_CPLUSPLUS20 + requires(simdutf::detail::indexes_into_byte_like<InputPtr> && + simdutf::detail::index_assignable_from_char<OutputPtr>) +#endif +simdutf_constexpr23 full_result convert_safe_with_details_constexpr( + InputPtr data, size_t len, OutputPtr utf8_output, size_t utf8_len) { + const size_t output_count = + convert_safe_constexpr(data, len, utf8_output, utf8_len); + size_t input_count = 0; + size_t counted_output = 0; + while (input_count < len) { + const size_t width = + uint8_t(data[input_count]) < uint8_t(0x80) ? size_t(1) : size_t(2); + if (counted_output + width > output_count) { + break; + } + input_count++; + counted_output += width; + } + return full_result(input_count == len ? error_code::SUCCESS + : error_code::OUTPUT_BUFFER_TOO_SMALL, + input_count, output_count); +} + template <typename InputPtr> #if SIMDUTF_CPLUSPLUS20 requires simdutf::detail::indexes_into_byte_like<InputPtr> diff -urN '--exclude=CVS' '--exclude=.cvsignore' '--exclude=.svn' '--exclude=.svnignore' old/simdutf-9.1.2/include/simdutf/scalar/utf16_to_utf8/utf16_to_utf8.h new/simdutf-9.2.0/include/simdutf/scalar/utf16_to_utf8/utf16_to_utf8.h --- old/simdutf-9.1.2/include/simdutf/scalar/utf16_to_utf8/utf16_to_utf8.h 2026-09-11 04:41:45.000000000 +0200 +++ new/simdutf-9.2.0/include/simdutf/scalar/utf16_to_utf8/utf16_to_utf8.h 2026-09-15 17:49:26.000000000 +0200 @@ -289,6 +289,71 @@ return utf8_output - start; } +template <endianness big_endian, typename InputPtr, typename OutputPtr> +#if SIMDUTF_CPLUSPLUS20 + requires(simdutf::detail::indexes_into_utf16<InputPtr> && + simdutf::detail::index_assignable_from_char<OutputPtr>) +#endif +simdutf_constexpr23 full_result convert_with_replacement_safe( + InputPtr data, size_t len, OutputPtr utf8_output, size_t utf8_len) { + if (len == 0) { + return full_result(error_code::SUCCESS, 0, 0); + } + if (utf8_len == 0) { + return full_result(error_code::OUTPUT_BUFFER_TOO_SMALL, 0, 0); + } + + size_t input_count = 0; + size_t output_count = 0; + while (input_count < len) { + full_result r = convert_with_errors<big_endian, true>( + data + input_count, len - input_count, utf8_output + output_count, + utf8_len - output_count); + input_count += r.input_count; + output_count += r.output_count; + + if (r.error == error_code::SUCCESS) { + return full_result(error_code::SUCCESS, input_count, output_count); + } + + if (r.error == error_code::OUTPUT_BUFFER_TOO_SMALL) { + if (utf8_len - output_count < 3) { + return full_result(r.error, input_count, output_count); + } + + uint16_t word = !match_system(big_endian) + ? u16_swap_bytes(data[input_count]) + : data[input_count]; + bool unpaired = (word & 0xfc00) == 0xdc00; + if ((word & 0xfc00) == 0xd800) { + if (input_count + 1 == len) { + unpaired = true; + } else { + const uint16_t next_word = !match_system(big_endian) + ? u16_swap_bytes(data[input_count + 1]) + : data[input_count + 1]; + unpaired = (next_word & 0xfc00) != 0xdc00; + } + } + if (!unpaired) { + return full_result(r.error, input_count, output_count); + } + } else if (r.error != error_code::SURROGATE) { + return full_result(r.error, input_count, output_count); + } + + if (utf8_len - output_count < 3) { + return full_result(error_code::OUTPUT_BUFFER_TOO_SMALL, input_count, + output_count); + } + utf8_output[output_count++] = char(0xef); + utf8_output[output_count++] = char(0xbf); + utf8_output[output_count++] = char(0xbd); + input_count++; + } + return full_result(error_code::SUCCESS, input_count, output_count); +} + } // namespace utf16_to_utf8 } // unnamed namespace } // namespace scalar diff -urN '--exclude=CVS' '--exclude=.cvsignore' '--exclude=.svn' '--exclude=.svnignore' old/simdutf-9.1.2/include/simdutf/simdutf_version.h new/simdutf-9.2.0/include/simdutf/simdutf_version.h --- old/simdutf-9.1.2/include/simdutf/simdutf_version.h 2026-09-11 04:41:45.000000000 +0200 +++ new/simdutf-9.2.0/include/simdutf/simdutf_version.h 2026-09-15 17:49:26.000000000 +0200 @@ -4,7 +4,7 @@ #define SIMDUTF_SIMDUTF_VERSION_H /** The version of simdutf being used (major.minor.revision) */ -#define SIMDUTF_VERSION "9.1.2" +#define SIMDUTF_VERSION "9.2.0" namespace simdutf { enum { @@ -15,11 +15,11 @@ /** * The minor version (major.MINOR.revision) of simdutf being used. */ - SIMDUTF_VERSION_MINOR = 1, + SIMDUTF_VERSION_MINOR = 2, /** * The revision (major.minor.REVISION) of simdutf being used. */ - SIMDUTF_VERSION_REVISION = 2 + SIMDUTF_VERSION_REVISION = 0 }; } // namespace simdutf diff -urN '--exclude=CVS' '--exclude=.cvsignore' '--exclude=.svn' '--exclude=.svnignore' old/simdutf-9.1.2/include/simdutf_c.h new/simdutf-9.2.0/include/simdutf_c.h --- old/simdutf-9.1.2/include/simdutf_c.h 2026-09-11 04:41:45.000000000 +0200 +++ new/simdutf-9.2.0/include/simdutf_c.h 2026-09-15 17:49:26.000000000 +0200 @@ -139,6 +139,8 @@ char *output); size_t simdutf_convert_latin1_to_utf8_safe(const char *input, size_t length, char *output, size_t utf8_len); +simdutf_full_result simdutf_convert_latin1_to_utf8_safe_with_details( + const char *input, size_t length, char *output, size_t utf8_len); size_t simdutf_convert_latin1_to_utf16le(const char *input, size_t length, char16_t *output); size_t simdutf_convert_latin1_to_utf16be(const char *input, size_t length, @@ -194,6 +196,8 @@ char *output); size_t simdutf_convert_utf16_to_utf8_safe(const char16_t *input, size_t length, char *output, size_t utf8_len); +simdutf_full_result simdutf_convert_utf16_to_utf8_safe_with_details( + const char16_t *input, size_t length, char *output, size_t utf8_len); size_t simdutf_convert_utf16_to_latin1(const char16_t *input, size_t length, char *output); size_t simdutf_convert_utf16le_to_latin1(const char16_t *input, size_t length, @@ -227,6 +231,8 @@ size_t simdutf_convert_utf16_to_utf8_with_replacement(const char16_t *input, size_t length, char *output); +simdutf_full_result simdutf_convert_utf16_to_utf8_with_replacement_safe( + const char16_t *input, size_t length, char *output, size_t utf8_len); size_t simdutf_convert_utf16le_to_utf8_with_replacement(const char16_t *input, size_t length, char *output); diff -urN '--exclude=CVS' '--exclude=.cvsignore' '--exclude=.svn' '--exclude=.svnignore' old/simdutf-9.1.2/scripts/release.py new/simdutf-9.2.0/scripts/release.py --- old/simdutf-9.1.2/scripts/release.py 2026-09-11 04:41:45.000000000 +0200 +++ new/simdutf-9.2.0/scripts/release.py 2026-09-15 17:49:26.000000000 +0200 @@ -159,10 +159,12 @@ print("Failed to run amalgamate") print("running doxygen") -cp = subprocess.run(["doxygen"], stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, cwd=maindir) # doesn't capture output - -if(cp.returncode != 0): - print("Failed to run doxygen") +try: + cp = subprocess.run(["doxygen"], stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, cwd=maindir) # doesn't capture output + if(cp.returncode != 0): + print("Failed to run doxygen") +except FileNotFoundError: + print("doxygen is not installed, skipping (the documentation workflow regenerates it)") readmefile = maindir + os.sep + "README.md" diff -urN '--exclude=CVS' '--exclude=.cvsignore' '--exclude=.svn' '--exclude=.svnignore' old/simdutf-9.1.2/src/arm64/arm_utf16fix.cpp new/simdutf-9.2.0/src/arm64/arm_utf16fix.cpp --- old/simdutf-9.1.2/src/arm64/arm_utf16fix.cpp 2026-09-11 04:41:45.000000000 +0200 +++ new/simdutf-9.2.0/src/arm64/arm_utf16fix.cpp 2026-09-15 17:49:26.000000000 +0200 @@ -1,9 +1,6 @@ /* - * Returns whether a vector of type uint8x16_t is not all zero. The input is - * always a combination of comparison masks (bytes equal to 0x00 or 0xff), so we - * can use the two-instruction test (shrn + fcmp) instead of a reduction - * followed by a costly move to a general-purpose register. + * Returns whether a vector of type uint8x16_t is not all zero. */ simdutf_really_inline bool veq_non_zero(uint8x16_t v) { return any_lane_set(vreinterpretq_u16_u8(v)); diff -urN '--exclude=CVS' '--exclude=.cvsignore' '--exclude=.svn' '--exclude=.svnignore' old/simdutf-9.1.2/src/implementation.cpp new/simdutf-9.2.0/src/implementation.cpp --- old/simdutf-9.1.2/src/implementation.cpp 2026-09-11 04:41:45.000000000 +0200 +++ new/simdutf-9.2.0/src/implementation.cpp 2026-09-15 17:49:26.000000000 +0200 @@ -2014,6 +2014,72 @@ } return r.output_count + (utf8_output - start); } + +simdutf_warn_unused full_result convert_utf16_to_utf8_safe_with_details( + const char16_t *buf, size_t len, char *utf8_output, + size_t utf8_len) noexcept { + if (len == 0) { + return full_result(error_code::SUCCESS, 0, 0); + } + size_t input_count = 0; + size_t output_count = 0; + // We might be able to go faster by first scanning the input buffer to + // determine how many char16_t characters we can read without exceeding the + // utf8_len. This is a one-pass algorithm that has the benefit of not + // requiring a first pass to determine the length. + while (true) { + // The worst case for convert_utf16_to_utf8 is when you go from 1 char16_t + // to 3 characters of UTF-8. So we can read at most utf8_len / 3 char16_t + // characters. + auto read_len = detail::min(len, utf8_len / 3); + if (read_len <= 16) { + break; + } + if (read_len < len) { + // If we have a high surrogate at the end of the buffer, we need to + // either read one more char16_t or backtrack. + if (scalar::utf16::high_surrogate(buf[read_len - 1])) { + read_len--; + } + } + if (read_len == 0) { + // If we cannot read anything, we are done. + break; + } + const result conversion_result = + simdutf::convert_utf16_to_utf8_with_errors(buf, read_len, utf8_output); + if (conversion_result.error != error_code::SUCCESS) { + const size_t valid_output_count = + simdutf::utf8_length_from_utf16(buf, conversion_result.count); + return full_result(conversion_result.error, + input_count + conversion_result.count, + output_count + valid_output_count); + } + + const size_t write_len = conversion_result.count; + utf8_output += write_len; + utf8_len -= write_len; + buf += read_len; + len -= read_len; + input_count += read_len; + output_count += write_len; + } + if (len == 0) { + return full_result(error_code::SUCCESS, input_count, output_count); + } + #if SIMDUTF_IS_BIG_ENDIAN + full_result r = + scalar::utf16_to_utf8::convert_with_errors<endianness::BIG, true>( + buf, len, utf8_output, utf8_len); + #else + full_result r = + scalar::utf16_to_utf8::convert_with_errors<endianness::LITTLE, true>( + buf, len, utf8_output, utf8_len); + #endif + r.input_count += input_count; + r.output_count += output_count; + return r; +} #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF16 #if SIMDUTF_FEATURE_UTF16 && SIMDUTF_FEATURE_LATIN1 @@ -2414,6 +2480,53 @@ #endif } +simdutf_warn_unused full_result convert_utf16_to_utf8_with_replacement_safe( + const char16_t *input, size_t length, char *utf8_output, + size_t utf8_len) noexcept { + size_t input_count = 0; + size_t output_count = 0; + while (input_count < length) { + const full_result r = convert_utf16_to_utf8_safe_with_details( + input + input_count, length - input_count, utf8_output + output_count, + utf8_len - output_count); + input_count += r.input_count; + output_count += r.output_count; + + if (r.error == error_code::SUCCESS) { + return full_result(error_code::SUCCESS, input_count, output_count); + } + if (r.error == error_code::OUTPUT_BUFFER_TOO_SMALL) { + #if SIMDUTF_IS_BIG_ENDIAN + full_result tail = + scalar::utf16_to_utf8::convert_with_replacement_safe<endianness::BIG>( + input + input_count, length - input_count, + utf8_output + output_count, utf8_len - output_count); + #else + full_result tail = scalar::utf16_to_utf8::convert_with_replacement_safe< + endianness::LITTLE>(input + input_count, length - input_count, + utf8_output + output_count, + utf8_len - output_count); + #endif + tail.input_count += input_count; + tail.output_count += output_count; + return tail; + } + if (r.error != error_code::SURROGATE) { + return full_result(r.error, input_count, output_count); + } + + if (utf8_len - output_count < 3) { + return full_result(error_code::OUTPUT_BUFFER_TOO_SMALL, input_count, + output_count); + } + utf8_output[output_count++] = char(0xef); + utf8_output[output_count++] = char(0xbf); + utf8_output[output_count++] = char(0xbd); + input_count++; + } + return full_result(error_code::SUCCESS, input_count, output_count); +} + simdutf_warn_unused size_t convert_utf16le_to_utf8_with_replacement( const char16_t *input, size_t length, char *utf8_buffer) noexcept { return get_default_implementation()->convert_utf16le_to_utf8_with_replacement( @@ -2522,9 +2635,11 @@ // moved to implementation.h // simdutf_warn_unused bool base64_ignorable(char input, -// base64_options options) noexcept +// base64_options options) +// noexcept // simdutf_warn_unused bool base64_ignorable(char16_t input, -// base64_options options) noexcept +// base64_options options) +// noexcept // simdutf_warn_unused bool base64_valid(char input, // base64_options options) noexcept // simdutf_warn_unused bool base64_valid(char16_t input, @@ -2589,6 +2704,36 @@ return utf8_output - start; } + +simdutf_warn_unused full_result convert_latin1_to_utf8_safe_with_details( + const char *buf, size_t len, char *utf8_output, size_t utf8_len) noexcept { + size_t input_count = 0; + size_t output_count = 0; + + while (true) { + // convert_latin1_to_utf8 will never write more than input length * 2 + auto read_len = detail::min(len, utf8_len >> 1); + if (read_len <= 16) { + break; + } + + const auto write_len = + simdutf::convert_latin1_to_utf8(buf, read_len, utf8_output); + + utf8_output += write_len; + utf8_len -= write_len; + buf += read_len; + len -= read_len; + input_count += read_len; + output_count += write_len; + } + + full_result r = scalar::latin1_to_utf8::convert_safe_with_details( + buf, len, utf8_output, utf8_len); + r.input_count += input_count; + r.output_count += output_count; + return r; +} #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_LATIN1 #if SIMDUTF_FEATURE_BASE64 diff -urN '--exclude=CVS' '--exclude=.cvsignore' '--exclude=.svn' '--exclude=.svnignore' old/simdutf-9.1.2/src/simdutf/arm64/simd.h new/simdutf-9.2.0/src/simdutf/arm64/simd.h --- old/simdutf-9.1.2/src/simdutf/arm64/simd.h 2026-09-11 04:41:45.000000000 +0200 +++ new/simdutf-9.2.0/src/simdutf/arm64/simd.h 2026-09-15 17:49:26.000000000 +0200 @@ -66,30 +66,17 @@ } // namespace #endif // SIMDUTF_REGULAR_VISUAL_STUDIO -// Returns true if any lane of `mask` is set. The argument *must* be the result -// of a lane-wise comparison, i.e. each byte must be either 0x00 or 0xff. The -// lane width of the comparison is irrelevant: byte and 32-bit masks should be -// reinterpreted with vreinterpretq_u16_u8 / vreinterpretq_u16_u32 by the -// caller. +// Returns true if any lane of `mask` is non-zero. // -// This compiles to two instructions (shrn + fcmp) and, unlike a reduction such -// as vmaxvq_u8, it never moves the value to a general-purpose register. Such a -// transfer has a latency of about 3 cycles on Apple hardware, on top of the 3 -// cycles of the reduction itself. -// -// Both steps rely on the input being a comparison mask. The narrowing shift -// keeps bits 4..11 of each 16-bit lane, so it only preserves 'is non-zero' when -// every byte is 0x00 or 0xff. The floating-point comparison is safe for the -// same reason: the only non-zero bit pattern that compares equal to 0.0 is -0.0 -// (0x8000000000000000), and the most significant byte of the narrowed value can -// only be 0x00, 0x0f, 0xf0 or 0xff. +// A 32-bit umaxv is used rather than vmaxvq_u8/u16, which is slower on some +// cores. A floating-point compare against 0.0 is avoided because flush-to-zero +// makes denormal bit patterns compare equal to zero. // // There is deliberately a single overload: Visual Studio defines every 128-bit // NEON type as the same union type, so overloading on uint8x16_t, uint16x8_t // and uint32x4_t does not compile there. simdutf_really_inline bool any_lane_set(const uint16x8_t mask) { - const uint8x8_t narrowed = vshrn_n_u16(mask, 4); - return vget_lane_f64(vreinterpret_f64_u8(narrowed), 0) != 0.0; + return vmaxvq_u32(vreinterpretq_u32_u16(mask)) != 0; } template <typename T> struct simd8; diff -urN '--exclude=CVS' '--exclude=.cvsignore' '--exclude=.svn' '--exclude=.svnignore' old/simdutf-9.1.2/src/simdutf/arm64/simd32-inl.h new/simdutf-9.2.0/src/simdutf/arm64/simd32-inl.h --- old/simdutf-9.1.2/src/simdutf/arm64/simd32-inl.h 2026-09-11 04:41:45.000000000 +0200 +++ new/simdutf-9.2.0/src/simdutf/arm64/simd32-inl.h 2026-09-15 17:49:26.000000000 +0200 @@ -58,7 +58,7 @@ simdutf_really_inline simd32(const uint32x4_t v) : value(v) {} // simd32<bool> is only ever produced by lane-wise comparisons (and bitwise - // combinations thereof), so the cheap any_lane_set is always applicable. + // combinations thereof), so any_lane_set is always applicable. simdutf_really_inline bool any() const { return any_lane_set(vreinterpretq_u16_u32(value)); } diff -urN '--exclude=CVS' '--exclude=.cvsignore' '--exclude=.svn' '--exclude=.svnignore' old/simdutf-9.1.2/src/simdutf_c.cpp new/simdutf-9.2.0/src/simdutf_c.cpp --- old/simdutf-9.1.2/src/simdutf_c.cpp 2026-09-11 04:41:45.000000000 +0200 +++ new/simdutf-9.2.0/src/simdutf_c.cpp 2026-09-15 17:49:26.000000000 +0200 @@ -8,6 +8,14 @@ return out; } +static simdutf_full_result to_c_full_result(const simdutf::full_result &r) { + simdutf_full_result out; + out.error = static_cast<simdutf_error_code>(r.error); + out.input_count = r.input_count; + out.output_count = r.output_count; + return out; +} + /* The C wrapper depends on the library features. Only expose the C API when all relevant feature is enabled. This helps the single-header generator to omit the C wrapper when features are @@ -167,6 +175,11 @@ char *output, size_t utf8_len) { return simdutf::convert_latin1_to_utf8_safe(input, length, output, utf8_len); } +simdutf_full_result simdutf_convert_latin1_to_utf8_safe_with_details( + const char *input, size_t length, char *output, size_t utf8_len) { + return to_c_full_result(simdutf::convert_latin1_to_utf8_safe_with_details( + input, length, output, utf8_len)); +} size_t simdutf_convert_latin1_to_utf16le(const char *input, size_t length, char16_t *output) { return simdutf::convert_latin1_to_utf16le(input, length, output); @@ -262,6 +275,11 @@ char *output, size_t utf8_len) { return simdutf::convert_utf16_to_utf8_safe(input, length, output, utf8_len); } +simdutf_full_result simdutf_convert_utf16_to_utf8_safe_with_details( + const char16_t *input, size_t length, char *output, size_t utf8_len) { + return to_c_full_result(simdutf::convert_utf16_to_utf8_safe_with_details( + input, length, output, utf8_len)); +} size_t simdutf_convert_utf16_to_latin1(const char16_t *input, size_t length, char *output) { return simdutf::convert_utf16_to_latin1(input, length, output); @@ -326,6 +344,11 @@ char *output) { return simdutf::convert_utf16_to_utf8_with_replacement(input, length, output); } +simdutf_full_result simdutf_convert_utf16_to_utf8_with_replacement_safe( + const char16_t *input, size_t length, char *output, size_t utf8_len) { + return to_c_full_result(simdutf::convert_utf16_to_utf8_with_replacement_safe( + input, length, output, utf8_len)); +} size_t simdutf_convert_utf16le_to_utf8_with_replacement(const char16_t *input, size_t length, char *output) { @@ -537,14 +560,6 @@ return to_c_result(r); } -static simdutf_full_result to_c_full_result(const simdutf::full_result &r) { - simdutf_full_result out; - out.error = static_cast<simdutf_error_code>(r.error); - out.input_count = r.input_count; - out.output_count = r.output_count; - return out; -} - simdutf_full_result simdutf_base64_to_binary_details( const char *input, size_t length, char *output, simdutf_base64_options options, diff -urN '--exclude=CVS' '--exclude=.cvsignore' '--exclude=.svn' '--exclude=.svnignore' old/simdutf-9.1.2/tests/convert_latin1_to_utf8_tests.cpp new/simdutf-9.2.0/tests/convert_latin1_to_utf8_tests.cpp --- old/simdutf-9.1.2/tests/convert_latin1_to_utf8_tests.cpp 2026-09-11 04:41:45.000000000 +0200 +++ new/simdutf-9.2.0/tests/convert_latin1_to_utf8_tests.cpp 2026-09-15 17:49:26.000000000 +0200 @@ -56,6 +56,62 @@ } } +TEST(convert_all_latin1_safe_with_details) { + std::vector<char> latin1(1024); + for (size_t i = 0; i < latin1.size(); i++) { + latin1[i] = i & 0xff; + } + const size_t utf8_length = + simdutf::utf8_length_from_latin1(latin1.data(), latin1.size()); + std::vector<char> expected(utf8_length); + ASSERT_EQUAL(simdutf::convert_latin1_to_utf8(latin1.data(), latin1.size(), + expected.data()), + utf8_length); + + for (size_t output_size = 0; output_size <= utf8_length; output_size++) { + std::vector<char> output(output_size); + std::vector<char> legacy_output(output_size); + const simdutf::full_result result = + simdutf::convert_latin1_to_utf8_safe_with_details( + latin1.data(), latin1.size(), output.data(), output.size()); + const size_t legacy_result = simdutf::convert_latin1_to_utf8_safe( + latin1.data(), latin1.size(), legacy_output.data(), + legacy_output.size()); + + size_t expected_input_count = 0; + size_t expected_output_count = 0; + while (expected_input_count < latin1.size()) { + const size_t width = uint8_t(latin1[expected_input_count]) < 0x80 ? 1 : 2; + if (expected_output_count + width > output_size) { + break; + } + expected_input_count++; + expected_output_count += width; + } + + ASSERT_EQUAL(result.input_count, expected_input_count); + ASSERT_EQUAL(result.output_count, expected_output_count); + ASSERT_EQUAL(legacy_result, result.output_count); + ASSERT_EQUAL(result.error, + expected_input_count == latin1.size() + ? simdutf::error_code::SUCCESS + : simdutf::error_code::OUTPUT_BUFFER_TOO_SMALL); + ASSERT_TRUE(std::equal(output.begin(), output.begin() + result.output_count, + expected.begin())); + ASSERT_TRUE(std::equal(legacy_output.begin(), + legacy_output.begin() + legacy_result, + output.begin())); + } +} + +TEST(convert_latin1_to_utf8_safe_with_details_empty) { + const simdutf::full_result result = + simdutf::convert_latin1_to_utf8_safe_with_details(nullptr, 0, nullptr, 0); + ASSERT_EQUAL(result.error, simdutf::error_code::SUCCESS); + ASSERT_EQUAL(result.input_count, 0); + ASSERT_EQUAL(result.output_count, 0); +} + #if SIMDUTF_CPLUSPLUS23 TEST(compile_time_utf8_length_from_latin1) { @@ -115,6 +171,19 @@ static_assert(output == expected); } } + +TEST(compile_time_convert_latin1_to_utf8_safe_with_details) { + using namespace simdutf::tests::helpers; + + constexpr auto result = []() { + constexpr auto input = "k\xF6ttbulle"_latin1; + CTString<char8_t, 2> output{}; + return simdutf::convert_latin1_to_utf8_safe_with_details(input, output); + }(); + static_assert(result.error == simdutf::OUTPUT_BUFFER_TOO_SMALL); + static_assert(result.input_count == 1); + static_assert(result.output_count == 1); +} #endif TEST_MAIN diff -urN '--exclude=CVS' '--exclude=.cvsignore' '--exclude=.svn' '--exclude=.svnignore' old/simdutf-9.1.2/tests/convert_utf16_to_utf8_safe_tests.cpp new/simdutf-9.2.0/tests/convert_utf16_to_utf8_safe_tests.cpp --- old/simdutf-9.1.2/tests/convert_utf16_to_utf8_safe_tests.cpp 2026-09-11 04:41:45.000000000 +0200 +++ new/simdutf-9.2.0/tests/convert_utf16_to_utf8_safe_tests.cpp 2026-09-15 17:49:26.000000000 +0200 @@ -64,6 +64,111 @@ ASSERT_TRUE(written <= 2); } +TEST(safe_with_details_mixed_width) { + const std::vector<char16_t> input{u'A', + char16_t(0x00e9), + char16_t(0x20ac), + char16_t(0xd83d), + char16_t(0xde00), + u'B'}; + const size_t utf8_length = + simdutf::utf8_length_from_utf16(input.data(), input.size()); + std::vector<char> expected(utf8_length); + ASSERT_EQUAL(simdutf::convert_utf16_to_utf8(input.data(), input.size(), + expected.data()), + utf8_length); + + for (size_t output_size = 0; output_size <= utf8_length; output_size++) { + std::vector<char> output(output_size); + std::vector<char> legacy_output(output_size); + const simdutf::full_result result = + simdutf::convert_utf16_to_utf8_safe_with_details( + input.data(), input.size(), output.data(), output.size()); + const size_t legacy_result = simdutf::convert_utf16_to_utf8_safe( + input.data(), input.size(), legacy_output.data(), legacy_output.size()); + + size_t expected_input_count = 0; + size_t expected_output_count = 0; + while (expected_input_count < input.size()) { + const uint16_t word = input[expected_input_count]; + size_t input_width = 1; + size_t output_width; + if (word < 0x80) { + output_width = 1; + } else if (word < 0x800) { + output_width = 2; + } else if (word < 0xd800 || word > 0xdfff) { + output_width = 3; + } else { + input_width = 2; + output_width = 4; + } + if (expected_output_count + output_width > output_size) { + break; + } + expected_input_count += input_width; + expected_output_count += output_width; + } + + ASSERT_EQUAL(result.input_count, expected_input_count); + ASSERT_EQUAL(result.output_count, expected_output_count); + ASSERT_EQUAL(legacy_result, result.output_count); + ASSERT_EQUAL(result.error, + expected_input_count == input.size() + ? simdutf::error_code::SUCCESS + : simdutf::error_code::OUTPUT_BUFFER_TOO_SMALL); + ASSERT_TRUE(std::equal(output.begin(), output.begin() + result.output_count, + expected.begin())); + ASSERT_TRUE(std::equal(legacy_output.begin(), + legacy_output.begin() + legacy_result, + output.begin())); + } +} + +TEST(safe_with_details_unpaired_surrogate) { + std::vector<char16_t> input(64, u'A'); + input.push_back(char16_t(0xdc00)); + input.push_back(u'B'); + std::vector<char> output(128); + + const simdutf::full_result result = + simdutf::convert_utf16_to_utf8_safe_with_details( + input.data(), input.size(), output.data(), output.size()); + ASSERT_EQUAL(result.error, simdutf::error_code::SURROGATE); + ASSERT_EQUAL(result.input_count, 64); + ASSERT_EQUAL(result.output_count, 64); + ASSERT_EQUAL(simdutf::convert_utf16_to_utf8_safe( + input.data(), input.size(), output.data(), output.size()), + 0); + + const simdutf::full_result full_output = + simdutf::convert_utf16_to_utf8_safe_with_details( + input.data(), input.size(), output.data(), 64); + ASSERT_EQUAL(full_output.error, simdutf::error_code::OUTPUT_BUFFER_TOO_SMALL); + ASSERT_EQUAL(full_output.input_count, 64); + ASSERT_EQUAL(full_output.output_count, 64); +} + +TEST(safe_with_details_empty) { + const simdutf::full_result result = + simdutf::convert_utf16_to_utf8_safe_with_details(nullptr, 0, nullptr, 0); + ASSERT_EQUAL(result.error, simdutf::error_code::SUCCESS); + ASSERT_EQUAL(result.input_count, 0); + ASSERT_EQUAL(result.output_count, 0); +} + +TEST(safe_with_details_exact_fit_three_byte) { + constexpr size_t n = 17; + std::vector<char16_t> input(n, char16_t(0x20ac)); + std::vector<char> output(3 * n); + const simdutf::full_result result = + simdutf::convert_utf16_to_utf8_safe_with_details( + input.data(), input.size(), output.data(), output.size()); + ASSERT_EQUAL(result.error, simdutf::error_code::SUCCESS); + ASSERT_EQUAL(result.input_count, n); + ASSERT_EQUAL(result.output_count, 3 * n); +} + TEST(convert_pure_ASCII) { size_t counter = 0; auto generator = [&counter]() -> uint32_t { return counter++ & 0x7f; }; @@ -198,6 +303,18 @@ static_assert(expected.shrink<N>() == actual.shrink<N>()); } +TEST(compile_time_convert_utf16_to_utf8_safe_with_details) { + using namespace simdutf::tests::helpers; + constexpr auto result = []() { + constexpr auto input = u"\u00E9A"_utf16; + CTString<char8_t, 2> output{}; + return simdutf::convert_utf16_to_utf8_safe_with_details(input, output); + }(); + static_assert(result.error == simdutf::OUTPUT_BUFFER_TOO_SMALL); + static_assert(result.input_count == 1); + static_assert(result.output_count == 2); +} + #endif TEST_MAIN diff -urN '--exclude=CVS' '--exclude=.cvsignore' '--exclude=.svn' '--exclude=.svnignore' old/simdutf-9.1.2/tests/convert_utf16_to_utf8_with_replacement_tests.cpp new/simdutf-9.2.0/tests/convert_utf16_to_utf8_with_replacement_tests.cpp --- old/simdutf-9.1.2/tests/convert_utf16_to_utf8_with_replacement_tests.cpp 2026-09-11 04:41:45.000000000 +0200 +++ new/simdutf-9.2.0/tests/convert_utf16_to_utf8_with_replacement_tests.cpp 2026-09-15 17:49:26.000000000 +0200 @@ -15,6 +15,66 @@ // U+FFFD in UTF-8 is 0xEF 0xBF 0xBD constexpr char fffd_utf8[] = {char(0xef), char(0xbf), char(0xbd)}; +TEST(replacement_safe_reports_input_and_output_counts) { + std::vector<char16_t> input(32, u'A'); + const std::vector<char16_t> suffix{char16_t(0x00e9), char16_t(0x20ac), + char16_t(0xd83d), char16_t(0xde00), + char16_t(0xd800), u'B', + char16_t(0xdc00), u'C'}; + input.insert(input.end(), suffix.begin(), suffix.end()); + input.insert(input.end(), 32, u'Z'); + + const simdutf::result length_result = + simdutf::utf8_length_from_utf16_with_replacement(input.data(), + input.size()); + std::vector<char> expected(length_result.count); + ASSERT_EQUAL(simdutf::convert_utf16_to_utf8_with_replacement( + input.data(), input.size(), expected.data()), + expected.size()); + + for (size_t output_size = 0; output_size <= expected.size(); output_size++) { + std::vector<char> output(output_size); + const simdutf::full_result result = + simdutf::convert_utf16_to_utf8_with_replacement_safe( + input.data(), input.size(), output.data(), output.size()); + + size_t expected_input_count = 0; + size_t expected_output_count = 0; + while (expected_input_count < input.size()) { + const uint16_t word = input[expected_input_count]; + size_t input_width = 1; + size_t output_width; + if (word < 0x80) { + output_width = 1; + } else if (word < 0x800) { + output_width = 2; + } else if (word >= 0xd800 && word <= 0xdbff && + expected_input_count + 1 < input.size() && + input[expected_input_count + 1] >= 0xdc00 && + input[expected_input_count + 1] <= 0xdfff) { + input_width = 2; + output_width = 4; + } else { + output_width = 3; + } + if (expected_output_count + output_width > output_size) { + break; + } + expected_input_count += input_width; + expected_output_count += output_width; + } + + ASSERT_EQUAL(result.input_count, expected_input_count); + ASSERT_EQUAL(result.output_count, expected_output_count); + ASSERT_EQUAL(result.error, + expected_input_count == input.size() + ? simdutf::error_code::SUCCESS + : simdutf::error_code::OUTPUT_BUFFER_TOO_SMALL); + ASSERT_TRUE(std::equal(output.begin(), output.begin() + result.output_count, + expected.begin())); + } +} + // Test: valid UTF-16 should produce the same output as convert_utf16_to_utf8 TEST(valid_utf16le_roundtrip) { // ASCII + BMP characters @@ -448,6 +508,18 @@ static_assert(result == expected); } +TEST(compile_time_convert_utf16_to_utf8_with_replacement_safe) { + using enum simdutf::endianness; + constexpr auto result = []() { + constexpr auto input = make_input_with_unpaired<NATIVE>(); + simdutf::tests::helpers::CTString<char, 4> output{}; + return simdutf::convert_utf16_to_utf8_with_replacement_safe(input, output); + }(); + static_assert(result.error == simdutf::OUTPUT_BUFFER_TOO_SMALL); + static_assert(result.input_count == 2); + static_assert(result.output_count == 4); +} + #endif // SIMDUTF_CPLUSPLUS23 TEST_MAIN diff -urN '--exclude=CVS' '--exclude=.cvsignore' '--exclude=.svn' '--exclude=.svnignore' old/simdutf-9.1.2/tests/nostdlibcxx_c_api_test.c new/simdutf-9.2.0/tests/nostdlibcxx_c_api_test.c --- old/simdutf-9.1.2/tests/nostdlibcxx_c_api_test.c 2026-09-11 04:41:45.000000000 +0200 +++ new/simdutf-9.2.0/tests/nostdlibcxx_c_api_test.c 2026-09-15 17:49:26.000000000 +0200 @@ -52,6 +52,27 @@ EXPECT(m == 5); EXPECT(memcmp(back, hello_u8, 5) == 0); + /* --- Size-limited transcoding with input/output counts --- */ + simdutf_full_result latin1_details = + simdutf_convert_latin1_to_utf8_safe_with_details(ascii, 5, back, 3); + EXPECT(latin1_details.error == SIMDUTF_ERROR_OUTPUT_BUFFER_TOO_SMALL); + EXPECT(latin1_details.input_count == 3); + EXPECT(latin1_details.output_count == 3); + + simdutf_full_result utf16_details = + simdutf_convert_utf16_to_utf8_safe_with_details(u"Hello", 5, back, 3); + EXPECT(utf16_details.error == SIMDUTF_ERROR_OUTPUT_BUFFER_TOO_SMALL); + EXPECT(utf16_details.input_count == 3); + EXPECT(utf16_details.output_count == 3); + + char16_t broken_u16[2] = {(char16_t)0xd800, (char16_t)'A'}; + simdutf_full_result replacement_details = + simdutf_convert_utf16_to_utf8_with_replacement_safe(broken_u16, 2, back, + 3); + EXPECT(replacement_details.error == SIMDUTF_ERROR_OUTPUT_BUFFER_TOO_SMALL); + EXPECT(replacement_details.input_count == 1); + EXPECT(replacement_details.output_count == 3); + /* --- Base64 round-trip --- */ const char binary_in[] = "simdutf rocks!"; char b64[64]; diff -urN '--exclude=CVS' '--exclude=.cvsignore' '--exclude=.svn' '--exclude=.svnignore' old/simdutf-9.1.2/tests/span_tests.cpp new/simdutf-9.2.0/tests/span_tests.cpp --- old/simdutf-9.1.2/tests/span_tests.cpp 2026-09-11 04:41:45.000000000 +0200 +++ new/simdutf-9.2.0/tests/span_tests.cpp 2026-09-15 17:49:26.000000000 +0200 @@ -205,6 +205,16 @@ std::vector<char>>); } +TEST(convert_latin1_to_utf8_safe_with_details) { + const std::vector<char> input{'a', char(0xe9)}; + std::array<char, 2> output{}; + const simdutf::full_result result = + simdutf::convert_latin1_to_utf8_safe_with_details(input, output); + ASSERT_EQUAL(result.error, simdutf::OUTPUT_BUFFER_TOO_SMALL); + ASSERT_EQUAL(result.input_count, 1); + ASSERT_EQUAL(result.output_count, 1); +} + TEST(validate_utf32_with_errors) { std::array<char32_t, 3> data{1, 2, 3}; auto r1a = simdutf::validate_utf32_with_errors(data); @@ -245,6 +255,26 @@ auto r1b = simdutf::convert_utf16_to_utf8(std::as_const(input), output); } +TEST(convert_utf16_to_utf8_safe_with_details) { + const std::array<char16_t, 2> input{u'A', char16_t(0x00e9)}; + std::array<char, 2> output{}; + const simdutf::full_result result = + simdutf::convert_utf16_to_utf8_safe_with_details(input, output); + ASSERT_EQUAL(result.error, simdutf::OUTPUT_BUFFER_TOO_SMALL); + ASSERT_EQUAL(result.input_count, 1); + ASSERT_EQUAL(result.output_count, 1); +} + +TEST(convert_utf16_to_utf8_with_replacement_safe) { + const std::array<char16_t, 3> input{u'A', char16_t(0xd800), u'B'}; + std::array<char, 4> output{}; + const simdutf::full_result result = + simdutf::convert_utf16_to_utf8_with_replacement_safe(input, output); + ASSERT_EQUAL(result.error, simdutf::OUTPUT_BUFFER_TOO_SMALL); + ASSERT_EQUAL(result.input_count, 2); + ASSERT_EQUAL(result.output_count, 4); +} + TEST(convert_utf16_to_latin1) { std::array<char16_t, 4> input{}; std::string output; diff -urN '--exclude=CVS' '--exclude=.cvsignore' '--exclude=.svn' '--exclude=.svnignore' old/simdutf-9.1.2/tests/straight_c_test.c new/simdutf-9.2.0/tests/straight_c_test.c --- old/simdutf-9.1.2/tests/straight_c_test.c 2026-09-11 04:41:45.000000000 +0200 +++ new/simdutf-9.2.0/tests/straight_c_test.c 2026-09-15 17:49:26.000000000 +0200 @@ -107,6 +107,11 @@ char latin_out[8] = {0}; size_t latin_to_utf8 = simdutf_convert_latin1_to_utf8("abc", 3, latin_out); ASSERT_EQUAL_SIZE_T(latin_to_utf8, 3); + simdutf_full_result latin_details = + simdutf_convert_latin1_to_utf8_safe_with_details("abc", 3, latin_out, 2); + ASSERT_EQUAL_INT(latin_details.error, SIMDUTF_ERROR_OUTPUT_BUFFER_TOO_SMALL); + ASSERT_EQUAL_SIZE_T(latin_details.input_count, 2); + ASSERT_EQUAL_SIZE_T(latin_details.output_count, 2); /* prepare a UTF-16 sample */ char16_t u16[5] = {(char16_t)'h', (char16_t)'e', (char16_t)'l', (char16_t)'l', @@ -118,6 +123,20 @@ size_t safelen = simdutf_convert_utf16_to_utf8_safe(u16, 5, out8, sizeof(out8)); ASSERT_EQUAL_SIZE_T(safelen, u16len); + simdutf_full_result utf16_details = + simdutf_convert_utf16_to_utf8_safe_with_details(u16, 5, out8, 3); + ASSERT_EQUAL_INT(utf16_details.error, SIMDUTF_ERROR_OUTPUT_BUFFER_TOO_SMALL); + ASSERT_EQUAL_SIZE_T(utf16_details.input_count, 3); + ASSERT_EQUAL_SIZE_T(utf16_details.output_count, 3); + + char16_t broken_u16[2] = {(char16_t)0xd800, (char16_t)'A'}; + simdutf_full_result replacement_details = + simdutf_convert_utf16_to_utf8_with_replacement_safe(broken_u16, 2, out8, + 3); + ASSERT_EQUAL_INT(replacement_details.error, + SIMDUTF_ERROR_OUTPUT_BUFFER_TOO_SMALL); + ASSERT_EQUAL_SIZE_T(replacement_details.input_count, 1); + ASSERT_EQUAL_SIZE_T(replacement_details.output_count, 3); /* convert with errors */ simdutf_result cr = simdutf_convert_utf16_to_utf8_with_errors(u16, 5, out8); ++++++ simdutf.obsinfo ++++++ --- /var/tmp/diff_new_pack.uSKp4l/_old 2026-09-23 14:33:23.046407807 +0200 +++ /var/tmp/diff_new_pack.uSKp4l/_new 2026-09-23 14:33:23.050407974 +0200 @@ -1,5 +1,5 @@ name: simdutf -version: 9.1.2 -mtime: 1789094505 -commit: be33cbe34b2d222566fd88f1084f86771ed8ae4a +version: 9.2.0 +mtime: 1789487366 +commit: 8abc1d7a466bc882c2d72e1effd8661492db257c
