On 8/7/26 9:01 AM, Jorge Ramirez-Ortiz wrote: > UFS string descriptors are UTF-16 big-endian (JESD220), but > ufshcd_read_string_desc() fed the raw bytes to utf16_to_utf8(), which reads > host-endian code units, leaving dev_desc->model blank. > > Add an endian parameter to utf16_to_utf8() so the caller can specify the byte > order of the source, and pass UTF16_BIG_ENDIAN from the UFS driver, matching > the kernel's utf16s_to_utf8s(..., UTF16_BIG_ENDIAN). Existing callers keep > their current behaviour via UTF16_HOST_ENDIAN.
I'm not sure host-endian ever makes sense. UTF-16 is either going to be coming over a network or from a file, so needs to be big or little according to the protocol or defined file format. > > Signed-off-by: Jorge Ramirez-Ortiz <[email protected]> > --- > drivers/ufs/ufs-uclass.c | 7 ++++--- > include/charset.h | 17 ++++++++++++++++- > lib/charset.c | 15 ++++++++++++++- > lib/efi_loader/efi_file.c | 4 ++-- > 4 files changed, 36 insertions(+), 7 deletions(-) > > diff --git a/drivers/ufs/ufs-uclass.c b/drivers/ufs/ufs-uclass.c > index 6a51f337e47..5cde2ab70be 100644 > --- a/drivers/ufs/ufs-uclass.c > +++ b/drivers/ufs/ufs-uclass.c > @@ -1766,11 +1766,12 @@ static int ufshcd_read_string_desc(struct ufs_hba > *hba, int desc_index, > } > > /* > - * the descriptor contains string in UTF16 format > - * we need to convert to utf-8 so it can be displayed > + * the descriptor contains a big-endian UTF-16 string, convert > + * it to utf-8 so it can be displayed > */ > utf16_to_utf8(buff_ascii, > - (uint16_t *)&buf[QUERY_DESC_HDR_SIZE], ascii_len); > + (uint16_t *)&buf[QUERY_DESC_HDR_SIZE], ascii_len, > + UTF16_BIG_ENDIAN); > > /* replace non-printable or non-ASCII characters with spaces */ > for (i = 0; i < ascii_len; i++) > diff --git a/include/charset.h b/include/charset.h > index 348bad5883a..442cc44d077 100644 > --- a/include/charset.h > +++ b/include/charset.h > @@ -13,6 +13,19 @@ > > #define MAX_UTF8_PER_UTF16 3 > > +/** > + * enum utf16_endian - byte order of a UTF-16 string > + * > + * @UTF16_HOST_ENDIAN: code units are in host byte order > + * @UTF16_LITTLE_ENDIAN: code units are little-endian > + * @UTF16_BIG_ENDIAN: code units are big-endian > + */ > +enum utf16_endian { > + UTF16_HOST_ENDIAN, > + UTF16_LITTLE_ENDIAN, > + UTF16_BIG_ENDIAN, > +}; > + > /* > * codepage_437 - Unicode to codepage 437 translation table > */ > @@ -299,9 +312,11 @@ size_t u16_strlcat(u16 *dest, const u16 *src, size_t > count); > * @dest: the destination buffer to write the utf8 characters > * @src: the source utf16 string > * @size: the number of utf16 characters to convert > + * @endian: byte order of the code units in 'src' > * Return: the pointer to the first unwritten byte in 'dest' > */ > -uint8_t *utf16_to_utf8(uint8_t *dest, const uint16_t *src, size_t size); > +uint8_t *utf16_to_utf8(uint8_t *dest, const uint16_t *src, size_t size, > + enum utf16_endian endian); > > /** > * utf_to_cp() - translate Unicode code point to 8bit codepage > diff --git a/lib/charset.c b/lib/charset.c > index 182c92a50c4..e5861ba96f8 100644 > --- a/lib/charset.c > +++ b/lib/charset.c > @@ -11,6 +11,7 @@ > #include <efi_loader.h> > #include <errno.h> > #include <malloc.h> > +#include <asm/byteorder.h> > > /** > * codepage_437 - Unicode to codepage 437 translation table > @@ -458,13 +459,25 @@ size_t u16_strlcat(u16 *dest, const u16 *src, size_t > count) > } > > /* Convert UTF-16 to UTF-8. */ > -uint8_t *utf16_to_utf8(uint8_t *dest, const uint16_t *src, size_t size) > +uint8_t *utf16_to_utf8(uint8_t *dest, const uint16_t *src, size_t size, > + enum utf16_endian endian) > { > uint32_t code_high = 0; > > while (size--) { > uint32_t code = *src++; > > + switch (endian) { > + case UTF16_LITTLE_ENDIAN: > + code = le16_to_cpu(code); > + break; > + case UTF16_BIG_ENDIAN: > + code = be16_to_cpu(code); > + break; > + case UTF16_HOST_ENDIAN: > + break; > + } > + > if (code_high) { > if (code >= 0xDC00 && code <= 0xDFFF) { > /* Surrogate pair. */ > diff --git a/lib/efi_loader/efi_file.c b/lib/efi_loader/efi_file.c > index 19b43c4a625..b0faae2d716 100644 > --- a/lib/efi_loader/efi_file.c > +++ b/lib/efi_loader/efi_file.c > @@ -184,7 +184,7 @@ static struct efi_file_handle *file_open(struct > file_system *fs, > int flen = 0; > > if (file_name) { > - utf16_to_utf8((u8 *)f0, file_name, 1); > + utf16_to_utf8((u8 *)f0, file_name, 1, UTF16_HOST_ENDIAN); I'm assuming this should be UTF16_LITTLE_ENDIAN (because FAT file system). The bytes read from the file are not going to swap themselves on a big-endian system. > flen = u16_strlen(file_name); > } > > @@ -216,7 +216,7 @@ static struct efi_file_handle *file_open(struct > file_system *fs, > *p++ = '/'; > } > > - utf16_to_utf8((u8 *)p, file_name, flen); > + utf16_to_utf8((u8 *)p, file_name, flen, UTF16_HOST_ENDIAN); ditto > > if (sanitize_path(fh->path)) > goto error;
