[packages/coreutils] fmt: format 8-bit data instead of failing on it

arekm arekm at pld-linux.org
Tue Sep 15 08:38:33 CEST 2026


commit c5d7e9892d6bcd942f3716872002aa883456fe2a
Author: Arkadiusz Miśkiewicz <arekm at maven.pl>
Date:   Tue Sep 15 00:39:46 2026 +0200

    fmt: format 8-bit data instead of failing on it

 coreutils-fmt-wchars.patch | 287 ++++++++++++++++++++++++++++++++++-----------
 1 file changed, 217 insertions(+), 70 deletions(-)
---
diff --git a/coreutils-fmt-wchars.patch b/coreutils-fmt-wchars.patch
index 91a5ffe..d80a163 100644
--- a/coreutils-fmt-wchars.patch
+++ b/coreutils-fmt-wchars.patch
@@ -19,17 +19,28 @@
  #: src/fmt.c:298
  msgid ""
  "  -w, --width=WIDTH\n"
---- coreutils-9.10/src/fmt.c.orig	2026-01-30 14:31:27.000000000 +0100
-+++ coreutils-9.10/src/fmt.c	2026-02-09 20:56:04.510881761 +0100
-@@ -17,6 +17,7 @@
+--- coreutils-9.12/src/fmt.c.orig
++++ coreutils-9.12/src/fmt.c
+@@ -17,6 +17,8 @@
  /* Written by Ross Paterson <rap at doc.ic.ac.uk>.  */
  
  #include <config.h>
 +#include <wchar.h>
++#include <stdlib.h>
  #include <stdio.h>
  #include <sys/types.h>
  #include <getopt.h>
-@@ -38,7 +39,7 @@
+@@ -28,6 +30,9 @@
+ #include "c-ctype.h"
+ #include "system.h"
+ #include "fadvise.h"
++#include "ioblksize.h"
++#include "mbbuf.h"
++#include <uchar.h>
+ #include "xdectoint.h"
+ 
+ /* The official name of this program (e.g., no 'g' prefix).  */
+@@ -38,7 +43,7 @@
  /* The following parameters represent the program's idea of what is
     "best".  Adjust to taste, subject to the caveats given.  */
  
@@ -38,7 +49,7 @@
  #define WIDTH	75
  
  /* Prefer lines to be LEEWAY % shorter than the maximum width, giving
-@@ -50,7 +51,7 @@
+@@ -50,7 +55,7 @@
  #define DEF_INDENT 3
  
  /* Costs and bonuses are expressed as the equivalent departure from the
@@ -47,7 +58,7 @@
     cost of 50 means that it is as bad as a line 5 characters too short
     or too long.  The definition of SHORT_COST(n) should not be changed.
     However, EQUIV(n) may need tuning.  */
-@@ -77,11 +78,11 @@
+@@ -77,11 +82,11 @@ typedef long int COST;
  #define LINE_COST	EQUIV (70)
  
  /* Cost of breaking a line after the first word of a sentence, where
@@ -61,17 +72,77 @@
  #define ORPHAN_COST(n)	(EQUIV (150) / ((n) + 2))
  
  /* Bonus for breaking a line at the end of a sentence.  */
-@@ -113,11 +114,21 @@
+@@ -113,11 +118,81 @@ typedef long int COST;
  #define MAXWORDS	1000
  #define MAXCHARS	5000
  
 +/* Wide character support */
 +
++/* A byte that is not part of a valid character in this locale is carried
++   through the formatter in the surrogate range and written back unchanged,
++   so that 8-bit input gets formatted rather than rejected.  */
++enum { BYTE_ESCAPE = 0xDC00 };
++
++static bool
++is_escaped_byte (wint_t c)
++{
++  return BYTE_ESCAPE <= c && c < BYTE_ESCAPE + 0x100;
++}
++
 +static inline int
 +xwcwidth (wchar_t wc)
 +{
-+  int w = wcwidth (wc);
-+  return w < 0 ? 0 : w;
++  if (is_escaped_byte (wc))
++    return 1;
++  /* c32width, not wcwidth, to match how the input is decoded.  */
++  int w = c32width (wc);
++  if (0 <= w)
++    return w;
++  /* A unibyte locale decodes every byte, including ones it calls unprintable;
++     such a byte occupies one column, as it did before fmt knew about wide
++     characters at all.  */
++  return MB_CUR_MAX == 1 && wc < 0x100 ? 1 : 0;
++}
++
++static mbbuf_t input_buf;
++static char input_bytes[IO_BUFSIZE];
++
++/* Return the next character of F, or WEOF at end of input.  Reading goes
++   through mbbuf so that an invalid byte is reported rather than ending the
++   input, and so that the stream keeps its byte orientation.  */
++
++static wint_t
++fmt_getwc (MAYBE_UNUSED FILE *f)
++{
++  mcel_t g = mbbuf_get_char (&input_buf);
++  if (g.ch == MBBUF_EOF)
++    return WEOF;
++  return g.err ? BYTE_ESCAPE + g.ch : g.ch;
++}
++
++/* Write C to stdout, which stays byte oriented so that escaped bytes go out
++   exactly as they came in.  */
++
++static wint_t
++fmt_putwc (wint_t c)
++{
++  if (is_escaped_byte (c))
++    return putchar (c - BYTE_ESCAPE) == EOF ? WEOF : c;
++
++  static mbstate_t mbs;
++  char buf[MB_LEN_MAX];
++  size_t n = c32rtomb (buf, c, &mbs);
++  if (n == (size_t) -1)
++    {
++      /* In a unibyte locale every byte decodes to a character, including
++         bytes the locale cannot encode back; write such a one as it came.  */
++      if (MB_CUR_MAX == 1 && c < 0x100)
++        return putchar (c) == EOF ? WEOF : c;
++      return WEOF;
++    }
++  if (fwrite (buf, 1, n, stdout) < n)
++    return WEOF;
++  return c;
 +}
 +
  /* Extra ctype(3)-style macros.  */
@@ -86,7 +157,7 @@
  
  /* Size of a tab stop, for expansion on input and re-introduction on
     output.  */
-@@ -132,8 +152,9 @@
+@@ -132,8 +207,9 @@ struct Word
  
      /* Static attributes determined during input.  */
  
@@ -98,7 +169,7 @@
      int space;			/* the size of the following space */
      unsigned int paren:1;	/* starts with open paren */
      unsigned int period:1;	/* ends in [.?!])* */
-@@ -142,7 +163,7 @@
+@@ -142,7 +218,7 @@ struct Word
  
      /* The remaining fields are computed during the optimization.  */
  
@@ -107,7 +178,7 @@
      COST best_cost;		/* cost of best paragraph starting here */
      WORD *next_break;		/* break which achieves best_cost */
    };
-@@ -152,16 +173,16 @@
+@@ -152,16 +228,16 @@ struct Word
  static void set_prefix (char *p);
  static bool fmt (FILE *f, char const *);
  static bool get_paragraph (FILE *f);
@@ -130,7 +201,7 @@
  static void put_paragraph (WORD *finish);
  static void put_line (WORD *w, int indent);
  static void put_word (WORD *w);
-@@ -181,8 +202,11 @@
+@@ -181,8 +257,11 @@ static bool split;
  /* If true, don't preserve inter-word spacing (default false).  */
  static bool uniform;
  
@@ -143,7 +214,7 @@
  
  /* User-supplied maximum line width (default WIDTH).  The only output
     lines longer than this will each comprise a single word.  */
-@@ -190,14 +214,14 @@
+@@ -190,14 +269,14 @@ static int max_width = WIDTH;
  
  /* Values derived from the option values.  */
  
@@ -163,7 +234,7 @@
  
  /* The preferred width of text lines, set to LEEWAY % less than max_width.  */
  static int goal_width;
-@@ -212,10 +236,10 @@
+@@ -212,10 +291,10 @@ static int out_column;
  
  /* Space for the paragraph text -- longer paragraphs are handled neatly
     (cf. flush_paragraph()).  */
@@ -176,7 +247,7 @@
  
  /* The words of a paragraph -- longer paragraphs are handled neatly
     (cf. flush_paragraph()).  */
-@@ -247,16 +271,16 @@
+@@ -247,16 +326,16 @@ static int other_indent;
     prefix (next_prefix_indent).  See get_paragraph() and copy_rest().  */
  
  /* The last character read from the input file.  */
@@ -196,7 +267,7 @@
  
  void
  usage (int status)
-@@ -293,7 +317,11 @@
+@@ -293,7 +372,11 @@ The option -WIDTH is an abbreviated form
  "));
        oputs (_("\
    -u, --uniform-spacing\n\
@@ -209,7 +280,7 @@
  "));
        oputs (_("\
    -w, --width=WIDTH\n\
-@@ -321,6 +349,7 @@
+@@ -321,6 +404,7 @@ static struct option const long_options[
    {"split-only", no_argument, NULL, 's'},
    {"tagged-paragraph", no_argument, NULL, 't'},
    {"uniform-spacing", no_argument, NULL, 'u'},
@@ -217,7 +288,7 @@
    {"width", required_argument, NULL, 'w'},
    {"goal", required_argument, NULL, 'g'},
    {GETOPT_HELP_OPTION_DECL},
-@@ -345,6 +374,8 @@
+@@ -344,6 +428,8 @@ main (int argc, char **argv)
  
    atexit (close_stdout);
  
@@ -226,7 +297,7 @@
    if (argc > 1 && argv[1][0] == '-' && c_isdigit (argv[1][1]))
      {
        /* Old option syntax; a dash followed by one or more digits.  */
-@@ -360,7 +390,7 @@
+@@ -355,7 +441,7 @@ main (int argc, char **argv)
        argc--;
      }
  
@@ -235,7 +306,7 @@
                                   long_options, NULL))
           != -1)
      switch (optchar)
-@@ -388,6 +418,10 @@
+@@ -383,6 +469,10 @@ main (int argc, char **argv)
          uniform = true;
          break;
  
@@ -246,7 +317,7 @@
        case 'w':
          max_width_option = optarg;
          break;
-@@ -467,26 +501,32 @@
+@@ -462,26 +552,45 @@ main (int argc, char **argv)
  }
  
  /* Trim space from the front and back of the string P, yielding the prefix,
@@ -257,7 +328,6 @@
  set_prefix (char *p)
  {
 -  char *s;
-+  size_t len;
 +  wchar_t *s;
  
    prefix_lead_space = 0;
@@ -274,9 +344,23 @@
 -    s--;
 -  *s = '\0';
 -  prefix_length = s - p;
-+  len = mbsrtowcs (NULL, (const char **) &p, 0, NULL);
-+  prefix = xmalloc (len * sizeof (wchar_t));
-+  mbsrtowcs (prefix, (const char **) &p, len, NULL);
++
++  /* Convert byte by byte rather than with mbsrtowcs, which gives up on the
++     whole prefix as soon as one byte is not a character in this locale.
++     Each byte yields at most one wide character.  */
++  prefix = xnmalloc (strlen (p) + 1, sizeof *prefix);
++  s = prefix;
++  for (char const *b = p; *b; )
++    {
++      mcel_t g = mcel_scanz (b);
++      /* On an encoding error mcel leaves CH zero and puts the byte in ERR,
++         unlike mbbuf_get_char which fills CH in; take the byte from ERR so
++         that a prefix escapes exactly like the input does.  */
++      *s++ = g.err ? BYTE_ESCAPE + g.err : g.ch;
++      b += g.len;
++    }
++  *s = L'\0';
++
 +  for (s = prefix; *s; s++)
 +    prefix_full_width += xwcwidth (*s);
 +  prefix_width = prefix_full_width;
@@ -289,7 +373,15 @@
  }
  
  /* Read F and send formatted output to stdout.
-@@ -574,24 +614,24 @@
+@@ -493,6 +602,7 @@ static bool
+ fmt (FILE *f, char const *file)
+ {
+   fadvise (f, FADVISE_SEQUENTIAL);
++  mbbuf_init (&input_buf, input_bytes, sizeof input_bytes, f);
+   tabs = false;
+   other_indent = 0;
+   next_char = get_prefix (f);
+@@ -569,24 +679,24 @@ set_other_indent (bool same_paragraph)
  static bool
  get_paragraph (FILE *f)
  {
@@ -317,11 +409,11 @@
            return false;
          }
 -      putchar ('\n');
-+      putwchar (L'\n');
++      fmt_putwc (L'\n');
        c = get_prefix (f);
      }
  
-@@ -648,24 +688,24 @@
+@@ -643,24 +753,24 @@ get_paragraph (FILE *f)
     that failed to match the prefix.  In the latter, C is \n or EOF.
     Return the character (\n or EOF) ending the line.  */
  
@@ -339,25 +431,25 @@
 -        putchar (*s++);
 -      if (c != EOF && c != '\n')
 +      for (wchar_t const *s = prefix; out_column != in_column && *s; out_column++)
-+        putwchar (*s++);
++        fmt_putwc (*s++);
 +      if (c != WEOF && c != L'\n')
          put_space (in_column - out_column);
 -      if (c == EOF && in_column >= next_prefix_indent + prefix_length)
 -        putchar ('\n');
 +      if (c == WEOF && in_column >= next_prefix_indent + prefix_width)
-+        putwchar (L'\n');
++        fmt_putwc (L'\n');
      }
 -  while (c != '\n' && c != EOF)
 +  while (c != L'\n' && c != WEOF)
      {
 -      putchar (c);
 -      c = getc (f);
-+      putwchar (c);
-+      c = getwc (f);
++      fmt_putwc (c);
++      c = fmt_getwc (f);
      }
    return c;
  }
-@@ -675,11 +715,11 @@
+@@ -670,11 +780,11 @@ copy_rest (FILE *f, int c)
     otherwise false.  */
  
  static bool
@@ -372,7 +464,7 @@
  }
  
  /* Read a line from input file F, given first non-blank character C
-@@ -690,11 +730,11 @@
+@@ -685,11 +795,11 @@ same_para (int c)
  
     Return the first non-blank character of the next line.  */
  
@@ -387,7 +479,7 @@
    WORD *end_of_word;
  
    end_of_parabuf = &parabuf[MAXCHARS];
-@@ -706,6 +746,7 @@
+@@ -701,6 +811,7 @@ get_line (FILE *f, int c)
        /* Scan word.  */
  
        word_limit->text = wptr;
@@ -395,13 +487,13 @@
        do
          {
            if (wptr == end_of_parabuf)
-@@ -714,10 +755,12 @@
+@@ -709,10 +820,12 @@ get_line (FILE *f, int c)
                flush_paragraph ();
              }
            *wptr++ = c;
 -          c = getc (f);
 +          word_limit->width += xwcwidth (c);
-+          c = getwc (f);
++          c = fmt_getwc (f);
          }
 -      while (c != EOF && !c_isspace (c));
 -      in_column += word_limit->length = wptr - word_limit->text;
@@ -411,7 +503,7 @@
        check_punctuation (word_limit);
  
        /* Scan inter-word space.  */
-@@ -725,11 +768,11 @@
+@@ -720,11 +833,11 @@ get_line (FILE *f, int c)
        start = in_column;
        c = get_space (f, c);
        word_limit->space = in_column - start;
@@ -427,7 +519,7 @@
        if (word_limit == end_of_word)
          {
            set_other_indent (true);
-@@ -737,33 +780,33 @@
+@@ -732,33 +845,33 @@ get_line (FILE *f, int c)
          }
        word_limit++;
      }
@@ -449,7 +541,7 @@
    in_column = 0;
 -  c = get_space (f, getc (f));
 -  if (prefix_length == 0)
-+  c = get_space (f, getwc (f));
++  c = get_space (f, fmt_getwc (f));
 +  if (prefix_width == 0)
      next_prefix_indent = prefix_lead_space < in_column ?
        prefix_lead_space : in_column;
@@ -465,11 +557,11 @@
              return c;
            in_column++;
 -          c = getc (f);
-+          c = getwc (f);
++          c = fmt_getwc (f);
          }
        c = get_space (f, c);
      }
-@@ -773,21 +816,21 @@
+@@ -768,21 +881,21 @@ get_prefix (FILE *f)
  /* Read blank characters from input file F, starting with C, and keeping
     in_column up-to-date.  Return first non-blank character.  */
  
@@ -492,11 +584,11 @@
        else
          return c;
 -      c = getc (f);
-+      c = getwc (f);
++      c = fmt_getwc (f);
      }
  }
  
-@@ -796,9 +839,9 @@
+@@ -791,9 +904,9 @@ get_space (FILE *f, int c)
  static void
  check_punctuation (WORD *w)
  {
@@ -509,7 +601,7 @@
  
    w->paren = isopen (*start);
    w->punct = !! ispunct (fin);
-@@ -822,10 +865,10 @@
+@@ -817,10 +930,10 @@ flush_paragraph (void)
  
    if (word_limit == word)
      {
@@ -519,12 +611,12 @@
 -
 +      wchar_t *outptr;
 +      for (outptr = parabuf; outptr < wptr; outptr++)
-+        if (putwchar (*outptr) == WEOF)
++        if (fmt_putwc (*outptr) == WEOF)
 +          write_error ();
        wptr = parabuf;
        return;
      }
-@@ -857,7 +900,8 @@
+@@ -852,7 +965,8 @@ flush_paragraph (void)
    /* Copy text of words down to start of parabuf -- we use memmove because
       the source and target may overlap.  */
  
@@ -534,7 +626,7 @@
    shift = split_point->text - parabuf;
    wptr -= shift;
  
-@@ -881,53 +925,53 @@
+@@ -876,53 +990,53 @@ static void
  fmt_paragraph (void)
  {
    WORD *w;
@@ -604,7 +696,7 @@
  }
  
  /* Work around <https://gcc.gnu.org/PR109628>.  */
-@@ -957,33 +1001,33 @@
+@@ -952,33 +1066,33 @@ base_cost (WORD *this)
        else if ((this - 1)->punct)
          cost -= PUNCT_BONUS;
        else if (this > word + 1 && (this - 2)->final)
@@ -644,29 +736,30 @@
        cost += RAGGED_COST (n);
      }
    return cost;
-@@ -1010,8 +1054,8 @@
+@@ -1005,8 +1119,9 @@ put_line (WORD *w, int indent)
  
    out_column = 0;
    put_space (prefix_indent);
 -  fputs (prefix, stdout);
 -  out_column += prefix_length;
-+  fputws (prefix, stdout);
++  for (wchar_t const *p = prefix; *p != L'\0'; p++)
++    fmt_putwc (*p);
 +  out_column += prefix_width;
    put_space (indent - out_column);
  
    endline = w->next_break - 1;
-@@ -1021,8 +1065,8 @@
+@@ -1016,8 +1131,8 @@ put_line (WORD *w, int indent)
        put_space (w->space);
      }
    put_word (w);
 -  last_line_length = out_column;
 -  putchar ('\n');
 +  last_line_width = out_column;
-+  putwchar (L'\n');
++  fmt_putwc (L'\n');
  
    if (ferror (stdout))
      write_error ();
-@@ -1033,10 +1077,10 @@
+@@ -1028,10 +1143,10 @@ put_line (WORD *w, int indent)
  static void
  put_word (WORD *w)
  {
@@ -675,24 +768,24 @@
    for (int n = w->length; n != 0; n--)
 -    putchar (*s++);
 -  out_column += w->length;
-+    putwchar (*s++);
++    fmt_putwc (*s++);
 +  out_column += w->width;
  }
  
  /* Output to stdout SPACE spaces, or equivalent tabs.  */
-@@ -1053,13 +1097,13 @@
+@@ -1048,13 +1163,13 @@ put_space (int space)
        if (out_column + 1 < tab_target)
          while (out_column < tab_target)
            {
 -            putchar ('\t');
-+            putwchar (L'\t');
++            fmt_putwc (L'\t');
              out_column = (out_column / TABWIDTH + 1) * TABWIDTH;
            }
      }
    while (out_column < space_target)
      {
 -      putchar (' ');
-+      putwchar (L' ');
++      fmt_putwc (L' ');
        out_column++;
      }
  }
@@ -709,19 +802,73 @@
  @optItem{fmt,- at var{width},}
  @optItemx{fmt,-w, at w{ }@var{width}}
  @optItemx{fmt,--width,=@var{width}}
-The 8-bit-pfx case cannot pass with this patch applied: under LC_ALL=C a
-byte above 0x7f is not a valid character, getwc() fails with EILSEQ and
-fmt reports a read error where upstream simply formats the bytes.
-
 --- coreutils-9.12/tests/fmt/base.pl.orig
 +++ coreutils-9.12/tests/fmt/base.pl
-@@ -24,9 +24,6 @@
- 
- my @Tests =
-     (
--     ['8-bit-pfx', qw (-p 'ç'),
--      {IN=> "ça\nçb\n"},
--      {OUT=>"ça b\n"}],
+@@ -27,6 +27,10 @@ my @Tests =
+      ['8-bit-pfx', qw (-p 'ç'),
+       {IN=> "ça\nçb\n"},
+       {OUT=>"ça b\n"}],
++     # bytes that are not characters in this locale are formatted, not rejected
++     ['8-bit-data', '-w 20',
++      {IN=> "\xe7a\n\xf3b\n"},
++      {OUT=>"\xe7a \xf3b\n"}],
       ['wide-1', '-w 32768',
        {ERR => "fmt: invalid width: '32768': $limits->{ERANGE}\n"}, {EXIT => 1}],
       ['wide-2', '-w 2147483647',
+--- coreutils-9.12/tests/fmt/invalid-prefix.sh.orig
++++ coreutils-9.12/tests/fmt/invalid-prefix.sh
+@@ -0,0 +1,44 @@
++#!/bin/sh
++# Ensure a prefix works when it holds a byte that is not a character here.
++
++# Copyright (C) 2026 Free Software Foundation, Inc.
++
++# This program is free software: you can redistribute it and/or modify
++# it under the terms of the GNU General Public License as published by
++# the Free Software Foundation, either version 3 of the License, or
++# (at your option) any later version.
++
++# This program is distributed in the hope that it will be useful,
++# but WITHOUT ANY WARRANTY; without even the implied warranty of
++# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
++# GNU General Public License for more details.
++
++# You should have received a copy of the GNU General Public License
++# along with this program.  If not, see <https://www.gnu.org/licenses/>.
++
++. "${srcdir=.}/tests/init.sh"; path_prepend_ ./src
++print_ver_ fmt
++
++printf '\377a\n\377b\n' > in || framework_failure_
++printf '\377a b\n' > exp || framework_failure_
++prefix=$(printf '\377') || framework_failure_
++
++# 0xff is a character in a unibyte locale and an encoding error in a UTF-8 one;
++# either way the prefix has to match the same byte in the input.
++for loc in C $LOCALE_FR_UTF8; do
++  test -z "$loc" && continue
++  test "$loc" = none && continue
++
++  LC_ALL="$loc" fmt -p "$prefix" in > out || fail=1
++  compare exp out || fail=1
++done
++
++# A multi-byte prefix, for comparison.
++if test -n "$LOCALE_FR_UTF8" && test "$LOCALE_FR_UTF8" != none; then
++  printf '\303\247a\n\303\247b\n' > in2 || framework_failure_
++  printf '\303\247a b\n' > exp2 || framework_failure_
++  LC_ALL="$LOCALE_FR_UTF8" fmt -p "$(printf '\303\247')" in2 > out2 || fail=1
++  compare exp2 out2 || fail=1
++fi
++
++Exit $fail
+--- coreutils-9.12/tests/local.mk.orig
++++ coreutils-9.12/tests/local.mk
+@@ -259,6 +259,7 @@
+   tests/chgrp/posix-H.sh			\
+   tests/chgrp/recurse.sh			\
+   tests/fmt/base.pl				\
++  tests/fmt/invalid-prefix.sh			\
+   tests/fmt/goal-option.sh			\
+   tests/fmt/long-line.sh			\
+   tests/fmt/non-space.sh			\
================================================================

---- gitweb:

http://git.pld-linux.org/gitweb.cgi/packages/coreutils.git/commitdiff/78b5229b744c3c51a9b78f198ff200d34d19a1dc



More information about the pld-cvs-commit mailing list