In UTF-8 locales, -w (without --whitespace-delimited=trimmed) used the
slow cut_fields_mb_any path.  Route it through cut_fields_bytesearch
with a single forward scan for ASCII SP/TAB and c32issep multi-byte
blanks, with the same semantics as mcel_isblank.

Do not memchr2 the remaining line (or pre-scan a whole ASCII run to
its end) on each field lookup: both are quadratic in the field count
for long lines of only multi-byte blanks or only ASCII blanks.

Preserve only truly incomplete UTF-8 tails via
mbrtoc32 == (size_t) -2, so complete characters are not held across
refill (cf. mbbuf_fill responsiveness).  Invalid UTF-8 and
continuation-only input remain field data and always make progress.

Timings on this host (LC_ALL=C.UTF-8):

  (a+U+2003)*40000 + b, cut -w -f999999:  ~0.00s (was ~0.69s with
    suffix-wide memchr2; matches master order)
  (a+SP)*500000 + z, cut -w -f250000:     ~0.00s (was ~10s with
    ASCII-run pre-scan + memchr2)
  mostly-ASCII ~21MiB, cut -w -f1:        ~0.05s (master ~0.12s)

* src/cut.c (utf8_tail_hold, utf8_is_incomplete_prefix)
(find_blank_delimiter): New.
(find_field_terminator, cut_fields_bytesearch, cut_fields_ws): Use them.
* tests/cut/cut.pl: Add boundary, invalid, mixed-run, glyph+SP, and
U+3000 cases.
* tests/cut/w-utf8-responsive.sh: Test streaming complete UTF-8
responsiveness.
* tests/cut/w-utf8-pathological.sh: Guard quadratic ASCII and MB blank
scans; cover mixed runs, leading MB blanks, and non-blank MB content.
* tests/local.mk: Reference the new tests.
* NEWS: Mention the improvement.

Signed-off-by: Iván Ezequiel Rodriguez <[email protected]>
---
 NEWS                             |   6 ++
 src/cut.c                        | 150 ++++++++++++++++++++++++++++---
 tests/cut/cut.pl                 |  32 +++++++
 tests/cut/w-utf8-pathological.sh | 134 +++++++++++++++++++++++++++
 tests/cut/w-utf8-responsive.sh   |  71 +++++++++++++++
 tests/local.mk                   |   2 +
 6 files changed, 382 insertions(+), 13 deletions(-)
 create mode 100755 tests/cut/w-utf8-pathological.sh
 create mode 100755 tests/cut/w-utf8-responsive.sh

diff --git a/NEWS b/NEWS
index 17a6333d3..d68952854 100644
--- a/NEWS
+++ b/NEWS
@@ -95,6 +95,12 @@ GNU coreutils NEWS                                    -*- 
outline -*-
 
 ** Improvements
 
+  'cut -w' is much faster in UTF-8 locales when not using
+  '--whitespace-delimited=trimmed', by scanning for ASCII blanks and
+  Unicode blank separators in a single forward pass (avoiding quadratic
+  suffix rescans), while preserving incomplete multi-byte sequences
+  across buffer boundaries.
+
   When built with the configure option '--with-wtmpdb', invocations of
   'who /var/log/wtmp' and 'users /var/log/wtmp' use the wtmpdb database
   instead of the file /var/log/wtmp.  This makes them Y2038-safe.
diff --git a/src/cut.c b/src/cut.c
index 83f24244d..e8c8b3137 100644
--- a/src/cut.c
+++ b/src/cut.c
@@ -337,6 +337,8 @@ struct bytesearch_context
   bool at_eof;
   char *line_end;
   bool line_end_known;
+  /* Length of the blank delimiter found by find_blank_delimiter.  */
+  idx_t blank_delim_len;
 };
 
 static inline void
@@ -345,6 +347,94 @@ bytesearch_context_reset (struct bytesearch_context *ctx)
   ctx->mode = BYTESEARCH_FIELDS;
   ctx->line_end = NULL;
   ctx->line_end_known = false;
+  ctx->blank_delim_len = 1;
+}
+
+/* Return how many trailing bytes of BUF[0..LEN) form an incomplete UTF-8
+   sequence ((size_t) -2 from mbrtoc32) and must be retained for the next
+   refill.  Complete characters and encoding errors are not held, so a
+   writer that pauses after valid UTF-8 does not stall cut waiting for more
+   input (cf. mbbuf_fill responsiveness).  At most one mbrtoc32 call per
+   candidate start in the last MCEL_LEN_MAX bytes.  */
+static idx_t
+utf8_tail_hold (char const *buf, idx_t len)
+{
+  if (len == 0 || to_uchar (buf[len - 1]) < 0x80)
+    return 0;
+
+  idx_t max_try = MIN (len, (idx_t) MCEL_LEN_MAX);
+  for (idx_t hold = max_try; hold >= 1; hold--)
+    {
+      char const *p = buf + len - hold;
+      if (to_uchar (*p) < 0x80)
+        continue;
+
+      mbstate_t mbs;
+      mbszero (&mbs);
+      char32_t wc;
+      size_t n = mbrtoc32 (&wc, p, hold, &mbs);
+      if (n == (size_t) -2)
+        return hold;
+    }
+
+  return 0;
+}
+
+/* True if BUF[0..LEN) begins with an incomplete UTF-8 sequence.  */
+static bool
+utf8_is_incomplete_prefix (char const *buf, idx_t len)
+{
+  if (len == 0 || to_uchar (buf[0]) < 0x80)
+    return false;
+
+  mbstate_t mbs;
+  mbszero (&mbs);
+  char32_t wc;
+  return mbrtoc32 (&wc, buf, len, &mbs) == (size_t) -2;
+}
+
+/* Locate the next -w field delimiter in BUF of length LEN.
+   In UTF-8 treat ASCII SP/TAB and c32issep multi-byte blanks as
+   delimiters.  Set *DELIM_LEN to the delimiter width.  Return NULL if
+   none is found.
+
+   Use a single forward pass: do not call memchr2 over the remaining
+   line.  Pre-scanning an ASCII run to its end (or memchr2 of the full
+   suffix) before classifying bytes is quadratic in the field count for
+   long lines of only ASCII blanks or only multi-byte blanks.  */
+
+static char *
+find_blank_delimiter (char *buf, idx_t len, idx_t *delim_len)
+{
+  *delim_len = 1;
+
+  if (! is_utf8_charset ())
+    return memchr2 (buf, ' ', '\t', len);
+
+  char *q = buf;
+  char *end = buf + len;
+
+  while (q < end)
+    {
+      unsigned char c = to_uchar (*q);
+      if (c < 0x80)
+        {
+          if (c == ' ' || c == '\t')
+            return q;
+          q++;
+          continue;
+        }
+
+      mcel_t g = mcel_scan (q, end);
+      if (! g.err && c32issep (g.ch))
+        {
+          *delim_len = g.len;
+          return q;
+        }
+      q += g.err ? 1 : g.len;
+    }
+
+  return NULL;
 }
 
 struct mbfield_parser
@@ -674,9 +764,11 @@ find_field_terminator (char *buf, idx_t len,
 
   idx_t field_len = ctx->line_end ? ctx->line_end - buf : len;
 
-  char *field_end = (ctx->blank_delimited
-                     ? memchr2 (buf, ' ', '\t', field_len)
-                     : find_field_delim (buf, field_len));
+  char *field_end;
+  if (ctx->blank_delimited)
+    field_end = find_blank_delimiter (buf, field_len, &ctx->blank_delim_len);
+  else
+    field_end = find_field_delim (buf, field_len);
 
   if (field_end)
     {
@@ -1118,9 +1210,37 @@ cut_fields_bytesearch (FILE *stream)
 
           if (skip_blank_run)
             {
-              while (processed < n_avail && c_isblank (chunk[processed]))
-                processed++;
-              if (processed == n_avail)
+              bool held_incomplete = false;
+              while (processed < n_avail)
+                {
+                  unsigned char c = to_uchar (chunk[processed]);
+                  if (c == ' ' || c == '\t')
+                    {
+                      processed++;
+                      continue;
+                    }
+                  if (c < 0x80 || ! is_utf8_charset ())
+                    break;
+
+                  /* Hold only a truly incomplete UTF-8 prefix, not every
+                     short high-bit tail (complete chars must not stall).  */
+                  if (! search.at_eof
+                      && utf8_is_incomplete_prefix (chunk + processed,
+                                                    n_avail - processed))
+                    {
+                      held_incomplete = true;
+                      break;
+                    }
+
+                  mcel_t g = mcel_scan (chunk + processed, chunk + n_avail);
+                  if (! g.err && c32issep (g.ch))
+                    {
+                      processed += g.len;
+                      continue;
+                    }
+                  break;
+                }
+              if (processed == n_avail || held_incomplete)
                 break;
               skip_blank_run = false;
             }
@@ -1155,11 +1275,13 @@ cut_fields_bytesearch (FILE *stream)
           idx_t field_len = terminator ? terminator - (chunk + processed)
                                        : n_avail - processed;
 
-          if (terminator_kind == FIELD_DATA
-              && !search.at_eof
-              && !whitespace_delimited
-              && !field_delim_is_line_delim ())
-            field_len -= field_delim_overlap (chunk + processed, field_len);
+          if (terminator_kind == FIELD_DATA && !search.at_eof)
+            {
+              if (!whitespace_delimited && !field_delim_is_line_delim ())
+                field_len -= field_delim_overlap (chunk + processed, 
field_len);
+              else if (whitespace_delimited && is_utf8_charset ())
+                field_len -= utf8_tail_hold (chunk + processed, field_len);
+            }
 
           if (field_len || terminator)
             have_pending_line = true;
@@ -1184,7 +1306,8 @@ cut_fields_bytesearch (FILE *stream)
                   break;
                 }
 
-              processed += whitespace_delimited ? 1 : delim_mcel.len;
+              processed += (whitespace_delimited
+                            ? search.blank_delim_len : delim_mcel.len);
               handle_field_delimiter (&field_idx, buffer_first_field,
                                       &field_1_n_bytes,
                                       &found_any_selected_field, &write_field,
@@ -1225,7 +1348,8 @@ cut_fields_bytesearch (FILE *stream)
 static void
 cut_fields_ws (FILE *stream)
 {
-  if (MB_CUR_MAX <= 1 && !trim_outer_whitespace)
+  if (!trim_outer_whitespace
+      && (MB_CUR_MAX <= 1 || is_utf8_charset ()))
     cut_fields_bytesearch (stream);
   else
     cut_fields_mb_any (stream, true);
diff --git a/tests/cut/cut.pl b/tests/cut/cut.pl
index 7e4a7ef80..b4033da3e 100755
--- a/tests/cut/cut.pl
+++ b/tests/cut/cut.pl
@@ -369,6 +369,38 @@ if ($mb_locale ne 'C')
        {ENV => "LC_ALL=$mb_locale"}],
       ['mb-w-nodelim-1', '-w', '-f2', {IN=>"abc"}, {OUT=>"abc\n"},
        {ENV => "LC_ALL=$mb_locale"}],
+      # U+2003 split across IO_BUFSIZE (analogous to mb-delim-9).
+      ['mb-w-delim-boundary', '-w', '-f2',
+       {IN=>('a' x ($IO_BUFSIZE - 1)) . "\xe2\x80\x83b\n"}, {OUT=>"b\n"},
+       {ENV => "LC_ALL=$mb_locale"}],
+      # Whitespace run spanning the buffer boundary stays one delimiter.
+      ['mb-w-run-boundary', '-w', '-f2',
+       {IN=>('a' x ($IO_BUFSIZE - 1)) . " \xe2\x80\x83b\n"}, {OUT=>"b\n"},
+       {ENV => "LC_ALL=$mb_locale"}],
+      # Invalid UTF-8 is field data, not a delimiter.
+      ['mb-w-invalid', '-w', '-f1,2',
+       {IN=>"a\xffb\tc\n"}, {OUT=>"a\xffb\tc\n"},
+       {ENV => "LC_ALL=$mb_locale"}],
+      # Invalid lead then ASCII blank: blank remains a -w delimiter.
+      ['mb-w-invalid-lead-ws', '-w', '-f2',
+       {IN=>"a\xe2 b\n"}, {OUT=>"b\n"},
+       {ENV => "LC_ALL=$mb_locale"}],
+      # Continuation-only bytes are consumed as data (no hang / hold-all).
+      ['mb-w-cont-only', '-w', '-f1',
+       {IN=>("\x80" x 8) . "\n"}, {OUT=>("\x80" x 8) . "\n"},
+       {ENV => "LC_ALL=$mb_locale"}],
+      # Mixed blank run (ASCII + U+2003) is a single -w delimiter.
+      ['mb-w-mixed-run', '-w', '-f2',
+       {IN=>"a \t\xe2\x80\x83\t b\n"}, {OUT=>"b\n"},
+       {ENV => "LC_ALL=$mb_locale"}],
+      # SP after non-blank multi-byte glyph.
+      ['mb-w-sp-after-glyph', '-w', '-f1,2',
+       {IN=>"a\xe2\x9c\x93 b\n"}, {OUT=>"a\xe2\x9c\x93\tb\n"},
+       {ENV => "LC_ALL=$mb_locale"}],
+      # U+3000 ideographic space as -w delimiter.
+      ['mb-w-ideo-space', '-w', '-f2',
+       {IN=>"a\xe3\x80\x80b\n"}, {OUT=>"b\n"},
+       {ENV => "LC_ALL=$mb_locale"}],
 
       # --complement with multi-byte
       ['mb-compl-c1', '--complement', '-c1',
diff --git a/tests/cut/w-utf8-pathological.sh b/tests/cut/w-utf8-pathological.sh
new file mode 100755
index 000000000..50848f3ee
--- /dev/null
+++ b/tests/cut/w-utf8-pathological.sh
@@ -0,0 +1,134 @@
+#!/bin/sh
+# Ensure cut -w stays linear on many blank delimiters (ASCII or UTF-8).
+# Regresses suffix-wide memchr2 / ASCII-run pre-scans that are quadratic
+# in the field count (see also the -f999999 multi-byte case on the list).
+
+# Copyright (C) 2026 Free Software Foundation, Inc.
+
+# This program is free software: you can redistribute it and/or modify
+# it under the terms of the GNU General Public License as published by
+# the Free Software Foundation, either version 3 of the License, or
+# (at your option) any later version.
+
+# This program is distributed in the hope that it will be useful,
+# but WITHOUT ANY WARRANTY; without even the implied warranty of
+# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
+# GNU General Public License for more details.
+
+# You should have received a copy of the GNU General Public License
+# along with this program.  If not, see <https://www.gnu.org/licenses/>.
+
+. "${srcdir=.}/tests/init.sh"; path_prepend_ ./src
+print_ver_ cut
+
+# Force a UTF-8 locale for the -w fast path under test.
+for loc in C.UTF-8 en_US.UTF-8 "$LOCALE_FR_UTF8"; do
+  test "$loc" && test "$loc" != none || continue
+  LC_ALL=$loc locale charmap 2>/dev/null | grep -i utf-8 >/dev/null &&
+    { export LC_ALL=$loc; break; }
+done
+LC_ALL=${LC_ALL:-C} locale charmap 2>/dev/null | grep -i utf-8 >/dev/null ||
+  skip_ 'UTF-8 locale required'
+
+em=$(printf '\xe2\x80\x83')   # U+2003 EM SPACE
+ideo=$(printf '\xe3\x80\x80') # U+3000 IDEOGRAPHIC SPACE
+
+# Helper: N copies of "a"<SEP>, then "b\n".  Late -f must finish quickly.
+make_fields() {
+  local sep=$1 n=$2 out=$3
+  {
+    yes "a$sep" | head -n"$n" | tr -d '\n'
+    printf 'b\n'
+  } > "$out" || framework_failure_
+}
+
+check_late_field() {
+  local in=$1 n=$2
+  local last=$((n + 1)) mid=$((n / 2))
+
+  echo a > exp || framework_failure_
+  timeout 2 cut -w -f1 "$in" > out || fail=1
+  compare exp out || fail=1
+
+  echo a > exp || framework_failure_
+  timeout 2 cut -w -f"$mid" "$in" > out || fail=1
+  compare exp out || fail=1
+
+  echo b > exp || framework_failure_
+  timeout 2 cut -w -f"$last" "$in" > out || fail=1
+  compare exp out || fail=1
+
+  echo > exp || framework_failure_
+  timeout 2 cut -w -f999999 "$in" > out || fail=1
+  compare exp out || fail=1
+}
+
+# Multi-byte blanks only (maintainer recipe, smaller N for the harness).
+n=8000
+make_fields "$em" "$n" in-em
+check_late_field in-em "$n"
+
+# Different multi-byte blank (U+3000).
+make_fields "$ideo" "$n" in-ideo
+check_late_field in-ideo "$n"
+
+# Pure ASCII blanks: pre-scanning each ASCII run to EOL is also quadratic.
+n_ascii=50000
+make_fields ' ' "$n_ascii" in-ascii
+check_late_field in-ascii "$n_ascii"
+
+# Short alternating ASCII + EM SPACE.
+{
+  yes "ab$em" | head -n"$n" | tr -d '\n'
+  printf 'z\n'
+} > in-alt || framework_failure_
+echo z > exp || framework_failure_
+timeout 2 cut -w -f$((n + 1)) in-alt > out || fail=1
+compare exp out || fail=1
+
+# Mixed blank run collapses to one delimiter.
+printf 'a \t%s\t b\n' "$em" > mix || framework_failure_
+printf 'b\n' > exp || framework_failure_
+cut -w -f2 mix > out || fail=1
+compare exp out || fail=1
+
+# SP/TAB after a non-blank multi-byte glyph.
+printf 'a\xe2\x9c\x93 b\n' > mix2 || framework_failure_
+printf 'a\xe2\x9c\x93\n' > exp || framework_failure_
+cut -w -f1 mix2 > out || fail=1
+compare exp out || fail=1
+printf 'b\n' > exp || framework_failure_
+cut -w -f2 mix2 > out || fail=1
+compare exp out || fail=1
+
+# Non-blank MB must not split the field.
+printf 'a\xe2\x9c\x93b c\n' > mix3 || framework_failure_
+printf 'a\xe2\x9c\x93b\n' > exp || framework_failure_
+cut -w -f1 mix3 > out || fail=1
+compare exp out || fail=1
+
+# Leading / consecutive MB blanks are one delimiter run under -w
+# (leading blanks yield an empty field 1, same as ASCII -w).
+printf '%s%sx%sy\n' "$em" "$em" "$em" > mix4 || framework_failure_
+echo > exp || framework_failure_
+cut -w -f1 mix4 > out || fail=1
+compare exp out || fail=1
+printf 'x\n' > exp || framework_failure_
+cut -w -f2 mix4 > out || fail=1
+compare exp out || fail=1
+printf 'y\n' > exp || framework_failure_
+cut -w -f3 mix4 > out || fail=1
+compare exp out || fail=1
+
+# Oracle: UTF-8 -w fast path must match trimmed=off mb path semantics.
+# Compare against a reference built from the same binary with a locale
+# that still uses UTF-8... we compare selected fields on crafted lines
+# against expected strings already checked above; add random-ish lines
+# vs an independent reimplementation using tr/awk only for ASCII, and
+# vs explicit expectations for MB.
+printf 'p%sq%sr\n' "$em" "$ideo" > oracle || framework_failure_
+printf 'p\tq\tr\n' > exp || framework_failure_
+cut -w -f1,2,3 oracle > out || fail=1
+compare exp out || fail=1
+
+Exit $fail
diff --git a/tests/cut/w-utf8-responsive.sh b/tests/cut/w-utf8-responsive.sh
new file mode 100755
index 000000000..cb7418bfc
--- /dev/null
+++ b/tests/cut/w-utf8-responsive.sh
@@ -0,0 +1,71 @@
+#!/bin/sh
+# Ensure cut -w processes complete UTF-8 without waiting for more input.
+# Regresses a hold-too-much fast path that reintroduced the latency
+# fixed by commit 57c87043f (mbbuf_fill responsiveness).
+
+# Copyright (C) 2026 Free Software Foundation, Inc.
+
+# This program is free software: you can redistribute it and/or modify
+# it under the terms of the GNU General Public License as published by
+# the Free Software Foundation, either version 3 of the License, or
+# (at your option) any later version.
+
+# This program is distributed in the hope that it will be useful,
+# but WITHOUT ANY WARRANTY; without even the implied warranty of
+# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
+# GNU General Public License for more details.
+
+# You should have received a copy of the GNU General Public License
+# along with this program.  If not, see <https://www.gnu.org/licenses/>.
+
+. "${srcdir=.}/tests/init.sh"; path_prepend_ ./src
+print_ver_ cut stdbuf
+
+require_built_ stdbuf
+
+# Force a UTF-8 locale for the -w fast path under test.
+for loc in C.UTF-8 en_US.UTF-8 "$LOCALE_FR_UTF8"; do
+  test "$loc" && test "$loc" != none || continue
+  LC_ALL=$loc locale charmap 2>/dev/null | grep -i utf-8 >/dev/null &&
+    { export LC_ALL=$loc; break; }
+done
+LC_ALL=${LC_ALL:-C} locale charmap 2>/dev/null | grep -i utf-8 >/dev/null ||
+  skip_ 'UTF-8 locale required'
+
+mkfifo_or_skip_ fifo
+
+# Writer pauses after a complete multi-byte character.  With an over-eager
+# hold of any high-bit tail, cut would block in read() before emitting "é".
+# stdbuf -o0 makes the emission observable before the writer resumes.
+check_responsive()
+{
+  local delay="$1"
+  compare exp out >/dev/null 2>&1 ||
+    { sleep $delay; return 1; }
+}
+
+printf 'caf\xc3\xa9' > exp || framework_failure_
+
+stdbuf -o0 cut -w -f1 > out < fifo & pid=$!
+
+# Keep the fifo writer in this shell so cut does not see EOF yet.
+exec 3>fifo
+printf 'caf\xc3\xa9' >&3 || framework_failure_
+
+# Before sending newline/EOF, cut must already have written complete UTF-8.
+retry_delay_ check_responsive .1 6 ||
+  {
+    cat out
+    fail=1
+  }
+
+printf ' x\n' >&3 || framework_failure_
+exec 3>&-
+
+wait $pid || fail=1
+
+# Final line: field1 is "café\n" (space starts field 2; line ends).
+printf 'caf\xc3\xa9\n' > exp || framework_failure_
+compare exp out || fail=1
+
+Exit $fail
diff --git a/tests/local.mk b/tests/local.mk
index 33abb9d72..7496a4819 100644
--- a/tests/local.mk
+++ b/tests/local.mk
@@ -305,6 +305,8 @@ all_tests =                                 \
   tests/misc/coreutils.sh                      \
   tests/cut/cut.pl                             \
   tests/cut/mb-non-utf8.sh                     \
+  tests/cut/w-utf8-responsive.sh               \
+  tests/cut/w-utf8-pathological.sh             \
   tests/cut/bounded-memory.sh                  \
   tests/cut/cut-huge-range.sh                  \
   tests/wc/wc.pl                               \
-- 
2.43.0




Reply via email to