In UTF-8 locales, -w (without --whitespace-delimited=trimmed) used the
slow cut_fields_mb_any path.  Route it through cut_fields_bytesearch
with memchr2 for ASCII blanks plus a linear scan for c32issep
multibyte blanks, with the same semantics as mcel_isblank.

Preserve only truly incomplete UTF-8 tails via
mbrtoc32 == (size_t) -2, so complete characters are not held across
refill (cf. mbbuf_fill responsiveness).  Invalid UTF-8 and
continuation-only input remain field data and always make progress.

Timings on this host (LC_ALL=C.UTF-8, cut -w -f1):

  mostly-ASCII ~20MiB:  ~0.073s -> ~0.018s
  heavy UTF-8  ~136MiB: ~0.67s  -> ~0.41s

* src/cut.c (utf8_tail_hold, utf8_is_incomplete_prefix)
(find_blank_delimiter): New.
(find_field_terminator, cut_fields_bytesearch, cut_fields_ws): Use them.
* tests/cut/cut.pl: Add boundary, invalid-lead+WS, and
continuation-only cases.
* tests/cut/w-utf8-responsive.sh: Test streaming complete UTF-8
responsiveness.
* tests/local.mk: Reference the new test.
* NEWS: Mention the improvement.

Signed-off-by: Iván Ezequiel Rodriguez <[email protected]>
---
 NEWS                           |   5 ++
 src/cut.c                      | 157 ++++++++++++++++++++++++++++++---
 tests/cut/cut.pl               |  20 +++++
 tests/cut/w-utf8-responsive.sh |  71 +++++++++++++++
 tests/local.mk                 |   1 +
 5 files changed, 241 insertions(+), 13 deletions(-)
 create mode 100755 tests/cut/w-utf8-responsive.sh

diff --git a/NEWS b/NEWS
index 17a6333d3..8efe40e54 100644
--- a/NEWS
+++ b/NEWS
@@ -95,6 +95,11 @@ GNU coreutils NEWS                                    -*- 
outline -*-
 
 ** Improvements
 
+  'cut -w' is much faster in UTF-8 locales when not using
+  '--whitespace-delimited=trimmed', by combining memchr2 for ASCII blanks
+  with a linear scan for Unicode blank separators, while preserving
+  incomplete multi-byte sequences across buffer boundaries.
+
   When built with the configure option '--with-wtmpdb', invocations of
   'who /var/log/wtmp' and 'users /var/log/wtmp' use the wtmpdb database
   instead of the file /var/log/wtmp.  This makes them Y2038-safe.
diff --git a/src/cut.c b/src/cut.c
index 83f24244d..dc7dd8ec2 100644
--- a/src/cut.c
+++ b/src/cut.c
@@ -337,6 +337,8 @@ struct bytesearch_context
   bool at_eof;
   char *line_end;
   bool line_end_known;
+  /* Length of the blank delimiter found by find_blank_delimiter.  */
+  idx_t blank_delim_len;
 };
 
 static inline void
@@ -345,6 +347,101 @@ bytesearch_context_reset (struct bytesearch_context *ctx)
   ctx->mode = BYTESEARCH_FIELDS;
   ctx->line_end = NULL;
   ctx->line_end_known = false;
+  ctx->blank_delim_len = 1;
+}
+
+/* Return how many trailing bytes of BUF[0..LEN) form an incomplete UTF-8
+   sequence ((size_t) -2 from mbrtoc32) and must be retained for the next
+   refill.  Complete characters and encoding errors are not held, so a
+   writer that pauses after valid UTF-8 does not stall cut waiting for more
+   input (cf. mbbuf_fill responsiveness).  At most one mbrtoc32 call per
+   candidate start in the last MCEL_LEN_MAX bytes.  */
+static idx_t
+utf8_tail_hold (char const *buf, idx_t len)
+{
+  if (len == 0 || to_uchar (buf[len - 1]) < 0x80)
+    return 0;
+
+  idx_t max_try = MIN (len, (idx_t) MCEL_LEN_MAX);
+  for (idx_t hold = max_try; hold >= 1; hold--)
+    {
+      char const *p = buf + len - hold;
+      if (to_uchar (*p) < 0x80)
+        continue;
+
+      mbstate_t mbs;
+      mbszero (&mbs);
+      char32_t wc;
+      size_t n = mbrtoc32 (&wc, p, hold, &mbs);
+      if (n == (size_t) -2)
+        return hold;
+    }
+
+  return 0;
+}
+
+/* True if BUF[0..LEN) begins with an incomplete UTF-8 sequence.  */
+static bool
+utf8_is_incomplete_prefix (char const *buf, idx_t len)
+{
+  if (len == 0 || to_uchar (buf[0]) < 0x80)
+    return false;
+
+  mbstate_t mbs;
+  mbszero (&mbs);
+  char32_t wc;
+  return mbrtoc32 (&wc, buf, len, &mbs) == (size_t) -2;
+}
+
+/* Locate the next -w field delimiter in BUF of length LEN.
+   Prefer memchr2 for ASCII SP/TAB; in UTF-8 also treat c32issep
+   multi-byte blanks as delimiters.  Set *DELIM_LEN to the delimiter width.
+   Return NULL if none is found.  Scans in O(LEN).  */
+
+static char *
+find_blank_delimiter (char *buf, idx_t len, idx_t *delim_len)
+{
+  *delim_len = 1;
+
+  if (! is_utf8_charset ())
+    return memchr2 (buf, ' ', '\t', len);
+
+  char *q = buf;
+  char *end = buf + len;
+
+  while (q < end)
+    {
+      idx_t rem = end - q;
+      char *sp = memchr2 (q, ' ', '\t', rem);
+      char *limit = sp ? sp : end;
+
+      while (q < limit)
+        {
+          unsigned char c = to_uchar (*q);
+          if (c < 0x80)
+            {
+              q++;
+              continue;
+            }
+
+          mcel_t g = mcel_scan (q, end);
+          if (! g.err && c32issep (g.ch))
+            {
+              *delim_len = g.len;
+              return q;
+            }
+          q += g.err ? 1 : g.len;
+        }
+
+      if (sp)
+        {
+          *delim_len = 1;
+          return sp;
+        }
+      return NULL;
+    }
+
+  return NULL;
 }
 
 struct mbfield_parser
@@ -674,9 +771,11 @@ find_field_terminator (char *buf, idx_t len,
 
   idx_t field_len = ctx->line_end ? ctx->line_end - buf : len;
 
-  char *field_end = (ctx->blank_delimited
-                     ? memchr2 (buf, ' ', '\t', field_len)
-                     : find_field_delim (buf, field_len));
+  char *field_end;
+  if (ctx->blank_delimited)
+    field_end = find_blank_delimiter (buf, field_len, &ctx->blank_delim_len);
+  else
+    field_end = find_field_delim (buf, field_len);
 
   if (field_end)
     {
@@ -1118,9 +1217,37 @@ cut_fields_bytesearch (FILE *stream)
 
           if (skip_blank_run)
             {
-              while (processed < n_avail && c_isblank (chunk[processed]))
-                processed++;
-              if (processed == n_avail)
+              bool held_incomplete = false;
+              while (processed < n_avail)
+                {
+                  unsigned char c = to_uchar (chunk[processed]);
+                  if (c == ' ' || c == '\t')
+                    {
+                      processed++;
+                      continue;
+                    }
+                  if (c < 0x80 || ! is_utf8_charset ())
+                    break;
+
+                  /* Hold only a truly incomplete UTF-8 prefix, not every
+                     short high-bit tail (complete chars must not stall).  */
+                  if (! search.at_eof
+                      && utf8_is_incomplete_prefix (chunk + processed,
+                                                    n_avail - processed))
+                    {
+                      held_incomplete = true;
+                      break;
+                    }
+
+                  mcel_t g = mcel_scan (chunk + processed, chunk + n_avail);
+                  if (! g.err && c32issep (g.ch))
+                    {
+                      processed += g.len;
+                      continue;
+                    }
+                  break;
+                }
+              if (processed == n_avail || held_incomplete)
                 break;
               skip_blank_run = false;
             }
@@ -1155,11 +1282,13 @@ cut_fields_bytesearch (FILE *stream)
           idx_t field_len = terminator ? terminator - (chunk + processed)
                                        : n_avail - processed;
 
-          if (terminator_kind == FIELD_DATA
-              && !search.at_eof
-              && !whitespace_delimited
-              && !field_delim_is_line_delim ())
-            field_len -= field_delim_overlap (chunk + processed, field_len);
+          if (terminator_kind == FIELD_DATA && !search.at_eof)
+            {
+              if (!whitespace_delimited && !field_delim_is_line_delim ())
+                field_len -= field_delim_overlap (chunk + processed, 
field_len);
+              else if (whitespace_delimited && is_utf8_charset ())
+                field_len -= utf8_tail_hold (chunk + processed, field_len);
+            }
 
           if (field_len || terminator)
             have_pending_line = true;
@@ -1184,7 +1313,8 @@ cut_fields_bytesearch (FILE *stream)
                   break;
                 }
 
-              processed += whitespace_delimited ? 1 : delim_mcel.len;
+              processed += (whitespace_delimited
+                            ? search.blank_delim_len : delim_mcel.len);
               handle_field_delimiter (&field_idx, buffer_first_field,
                                       &field_1_n_bytes,
                                       &found_any_selected_field, &write_field,
@@ -1225,7 +1355,8 @@ cut_fields_bytesearch (FILE *stream)
 static void
 cut_fields_ws (FILE *stream)
 {
-  if (MB_CUR_MAX <= 1 && !trim_outer_whitespace)
+  if (!trim_outer_whitespace
+      && (MB_CUR_MAX <= 1 || is_utf8_charset ()))
     cut_fields_bytesearch (stream);
   else
     cut_fields_mb_any (stream, true);
diff --git a/tests/cut/cut.pl b/tests/cut/cut.pl
index 7e4a7ef80..89c4fe8de 100755
--- a/tests/cut/cut.pl
+++ b/tests/cut/cut.pl
@@ -369,6 +369,26 @@ if ($mb_locale ne 'C')
        {ENV => "LC_ALL=$mb_locale"}],
       ['mb-w-nodelim-1', '-w', '-f2', {IN=>"abc"}, {OUT=>"abc\n"},
        {ENV => "LC_ALL=$mb_locale"}],
+      # U+2003 split across IO_BUFSIZE (analogous to mb-delim-9).
+      ['mb-w-delim-boundary', '-w', '-f2',
+       {IN=>('a' x ($IO_BUFSIZE - 1)) . "\xe2\x80\x83b\n"}, {OUT=>"b\n"},
+       {ENV => "LC_ALL=$mb_locale"}],
+      # Whitespace run spanning the buffer boundary stays one delimiter.
+      ['mb-w-run-boundary', '-w', '-f2',
+       {IN=>('a' x ($IO_BUFSIZE - 1)) . " \xe2\x80\x83b\n"}, {OUT=>"b\n"},
+       {ENV => "LC_ALL=$mb_locale"}],
+      # Invalid UTF-8 is field data, not a delimiter.
+      ['mb-w-invalid', '-w', '-f1,2',
+       {IN=>"a\xffb\tc\n"}, {OUT=>"a\xffb\tc\n"},
+       {ENV => "LC_ALL=$mb_locale"}],
+      # Invalid lead then ASCII blank: blank remains a -w delimiter.
+      ['mb-w-invalid-lead-ws', '-w', '-f2',
+       {IN=>"a\xe2 b\n"}, {OUT=>"b\n"},
+       {ENV => "LC_ALL=$mb_locale"}],
+      # Continuation-only bytes are consumed as data (no hang / hold-all).
+      ['mb-w-cont-only', '-w', '-f1',
+       {IN=>("\x80" x 8) . "\n"}, {OUT=>("\x80" x 8) . "\n"},
+       {ENV => "LC_ALL=$mb_locale"}],
 
       # --complement with multi-byte
       ['mb-compl-c1', '--complement', '-c1',
diff --git a/tests/cut/w-utf8-responsive.sh b/tests/cut/w-utf8-responsive.sh
new file mode 100755
index 000000000..cb7418bfc
--- /dev/null
+++ b/tests/cut/w-utf8-responsive.sh
@@ -0,0 +1,71 @@
+#!/bin/sh
+# Ensure cut -w processes complete UTF-8 without waiting for more input.
+# Regresses a hold-too-much fast path that reintroduced the latency
+# fixed by commit 57c87043f (mbbuf_fill responsiveness).
+
+# Copyright (C) 2026 Free Software Foundation, Inc.
+
+# This program is free software: you can redistribute it and/or modify
+# it under the terms of the GNU General Public License as published by
+# the Free Software Foundation, either version 3 of the License, or
+# (at your option) any later version.
+
+# This program is distributed in the hope that it will be useful,
+# but WITHOUT ANY WARRANTY; without even the implied warranty of
+# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
+# GNU General Public License for more details.
+
+# You should have received a copy of the GNU General Public License
+# along with this program.  If not, see <https://www.gnu.org/licenses/>.
+
+. "${srcdir=.}/tests/init.sh"; path_prepend_ ./src
+print_ver_ cut stdbuf
+
+require_built_ stdbuf
+
+# Force a UTF-8 locale for the -w fast path under test.
+for loc in C.UTF-8 en_US.UTF-8 "$LOCALE_FR_UTF8"; do
+  test "$loc" && test "$loc" != none || continue
+  LC_ALL=$loc locale charmap 2>/dev/null | grep -i utf-8 >/dev/null &&
+    { export LC_ALL=$loc; break; }
+done
+LC_ALL=${LC_ALL:-C} locale charmap 2>/dev/null | grep -i utf-8 >/dev/null ||
+  skip_ 'UTF-8 locale required'
+
+mkfifo_or_skip_ fifo
+
+# Writer pauses after a complete multi-byte character.  With an over-eager
+# hold of any high-bit tail, cut would block in read() before emitting "é".
+# stdbuf -o0 makes the emission observable before the writer resumes.
+check_responsive()
+{
+  local delay="$1"
+  compare exp out >/dev/null 2>&1 ||
+    { sleep $delay; return 1; }
+}
+
+printf 'caf\xc3\xa9' > exp || framework_failure_
+
+stdbuf -o0 cut -w -f1 > out < fifo & pid=$!
+
+# Keep the fifo writer in this shell so cut does not see EOF yet.
+exec 3>fifo
+printf 'caf\xc3\xa9' >&3 || framework_failure_
+
+# Before sending newline/EOF, cut must already have written complete UTF-8.
+retry_delay_ check_responsive .1 6 ||
+  {
+    cat out
+    fail=1
+  }
+
+printf ' x\n' >&3 || framework_failure_
+exec 3>&-
+
+wait $pid || fail=1
+
+# Final line: field1 is "café\n" (space starts field 2; line ends).
+printf 'caf\xc3\xa9\n' > exp || framework_failure_
+compare exp out || fail=1
+
+Exit $fail
diff --git a/tests/local.mk b/tests/local.mk
index 33abb9d72..7a99ecf8a 100644
--- a/tests/local.mk
+++ b/tests/local.mk
@@ -305,6 +305,7 @@ all_tests =                                 \
   tests/misc/coreutils.sh                      \
   tests/cut/cut.pl                             \
   tests/cut/mb-non-utf8.sh                     \
+  tests/cut/w-utf8-responsive.sh               \
   tests/cut/bounded-memory.sh                  \
   tests/cut/cut-huge-range.sh                  \
   tests/wc/wc.pl                               \
-- 
2.43.0




Reply via email to