On 8/7/26 9:01 AM, Jorge Ramirez-Ortiz wrote:
> UFS string descriptors are UTF-16 big-endian (JESD220), but
> ufshcd_read_string_desc() fed the raw bytes to utf16_to_utf8(), which reads
> host-endian code units, leaving dev_desc->model blank.
> 
> Add an endian parameter to utf16_to_utf8() so the caller can specify the byte
> order of the source, and pass UTF16_BIG_ENDIAN from the UFS driver, matching
> the kernel's utf16s_to_utf8s(..., UTF16_BIG_ENDIAN). Existing callers keep
> their current behaviour via UTF16_HOST_ENDIAN.

I'm not sure host-endian ever makes sense. UTF-16 is either going to be
coming over a network or from a file, so needs to be big or little according
to the protocol or defined file format.

> 
> Signed-off-by: Jorge Ramirez-Ortiz <[email protected]>
> ---
>  drivers/ufs/ufs-uclass.c  |  7 ++++---
>  include/charset.h         | 17 ++++++++++++++++-
>  lib/charset.c             | 15 ++++++++++++++-
>  lib/efi_loader/efi_file.c |  4 ++--
>  4 files changed, 36 insertions(+), 7 deletions(-)
> 
> diff --git a/drivers/ufs/ufs-uclass.c b/drivers/ufs/ufs-uclass.c
> index 6a51f337e47..5cde2ab70be 100644
> --- a/drivers/ufs/ufs-uclass.c
> +++ b/drivers/ufs/ufs-uclass.c
> @@ -1766,11 +1766,12 @@ static int ufshcd_read_string_desc(struct ufs_hba 
> *hba, int desc_index,
>               }
>  
>               /*
> -              * the descriptor contains string in UTF16 format
> -              * we need to convert to utf-8 so it can be displayed
> +              * the descriptor contains a big-endian UTF-16 string, convert
> +              * it to utf-8 so it can be displayed
>                */
>               utf16_to_utf8(buff_ascii,
> -                           (uint16_t *)&buf[QUERY_DESC_HDR_SIZE], ascii_len);
> +                           (uint16_t *)&buf[QUERY_DESC_HDR_SIZE], ascii_len,
> +                           UTF16_BIG_ENDIAN);
>  
>               /* replace non-printable or non-ASCII characters with spaces */
>               for (i = 0; i < ascii_len; i++)
> diff --git a/include/charset.h b/include/charset.h
> index 348bad5883a..442cc44d077 100644
> --- a/include/charset.h
> +++ b/include/charset.h
> @@ -13,6 +13,19 @@
>  
>  #define MAX_UTF8_PER_UTF16 3
>  
> +/**
> + * enum utf16_endian - byte order of a UTF-16 string
> + *
> + * @UTF16_HOST_ENDIAN:       code units are in host byte order
> + * @UTF16_LITTLE_ENDIAN:     code units are little-endian
> + * @UTF16_BIG_ENDIAN:        code units are big-endian
> + */
> +enum utf16_endian {
> +     UTF16_HOST_ENDIAN,
> +     UTF16_LITTLE_ENDIAN,
> +     UTF16_BIG_ENDIAN,
> +};
> +
>  /*
>   * codepage_437 - Unicode to codepage 437 translation table
>   */
> @@ -299,9 +312,11 @@ size_t u16_strlcat(u16 *dest, const u16 *src, size_t 
> count);
>   * @dest:    the destination buffer to write the utf8 characters
>   * @src:     the source utf16 string
>   * @size:    the number of utf16 characters to convert
> + * @endian:  byte order of the code units in 'src'
>   * Return:   the pointer to the first unwritten byte in 'dest'
>   */
> -uint8_t *utf16_to_utf8(uint8_t *dest, const uint16_t *src, size_t size);
> +uint8_t *utf16_to_utf8(uint8_t *dest, const uint16_t *src, size_t size,
> +                    enum utf16_endian endian);
>  
>  /**
>   * utf_to_cp() - translate Unicode code point to 8bit codepage
> diff --git a/lib/charset.c b/lib/charset.c
> index 182c92a50c4..e5861ba96f8 100644
> --- a/lib/charset.c
> +++ b/lib/charset.c
> @@ -11,6 +11,7 @@
>  #include <efi_loader.h>
>  #include <errno.h>
>  #include <malloc.h>
> +#include <asm/byteorder.h>
>  
>  /**
>   * codepage_437 - Unicode to codepage 437 translation table
> @@ -458,13 +459,25 @@ size_t u16_strlcat(u16 *dest, const u16 *src, size_t 
> count)
>  }
>  
>  /* Convert UTF-16 to UTF-8.  */
> -uint8_t *utf16_to_utf8(uint8_t *dest, const uint16_t *src, size_t size)
> +uint8_t *utf16_to_utf8(uint8_t *dest, const uint16_t *src, size_t size,
> +                    enum utf16_endian endian)
>  {
>       uint32_t code_high = 0;
>  
>       while (size--) {
>               uint32_t code = *src++;
>  
> +             switch (endian) {
> +             case UTF16_LITTLE_ENDIAN:
> +                     code = le16_to_cpu(code);
> +                     break;
> +             case UTF16_BIG_ENDIAN:
> +                     code = be16_to_cpu(code);
> +                     break;
> +             case UTF16_HOST_ENDIAN:
> +                     break;
> +             }
> +
>               if (code_high) {
>                       if (code >= 0xDC00 && code <= 0xDFFF) {
>                               /* Surrogate pair.  */
> diff --git a/lib/efi_loader/efi_file.c b/lib/efi_loader/efi_file.c
> index 19b43c4a625..b0faae2d716 100644
> --- a/lib/efi_loader/efi_file.c
> +++ b/lib/efi_loader/efi_file.c
> @@ -184,7 +184,7 @@ static struct efi_file_handle *file_open(struct 
> file_system *fs,
>       int flen = 0;
>  
>       if (file_name) {
> -             utf16_to_utf8((u8 *)f0, file_name, 1);
> +             utf16_to_utf8((u8 *)f0, file_name, 1, UTF16_HOST_ENDIAN);

I'm assuming this should be UTF16_LITTLE_ENDIAN (because FAT file system).
The bytes read from the file are not going to swap themselves on a
big-endian system.

>               flen = u16_strlen(file_name);
>       }
>  
> @@ -216,7 +216,7 @@ static struct efi_file_handle *file_open(struct 
> file_system *fs,
>                       *p++ = '/';
>               }
>  
> -             utf16_to_utf8((u8 *)p, file_name, flen);
> +             utf16_to_utf8((u8 *)p, file_name, flen, UTF16_HOST_ENDIAN);

ditto

>  
>               if (sanitize_path(fh->path))
>                       goto error;

Reply via email to