From: Robert Dubner <[email protected]>
Date: Wed, 12 Aug 2026 18:38:20 -0400
Subject: [PATCH] cobol: Refactor the use of cbl_char_t for big-endian
architecture.

One of the features of GCOBOL is that it can function with the
-fexec-charset set to ASCII, EBCDIC, UTF16LE, UTF16BE, UTF32LE, or
UTF32BE.  When I recently started work on getting the compiler to work
properly on a big-endian IBM System Z, I realized I had built in a lot
of little-endian assumptions that got in the way of handling big- and
little-endian character sets on a big-endian machine.

These changes eliminate some of those assumptions, so that the 32-bit
cbl_char_t means the same thing on both big- and little-endian
architectures.

gcc/cobol/ChangeLog:

        * genapi.cc (get_level_88_domain): Expanded comment.
        (create_and_call): Handle CALL USING BY VALUE ZERO.
        * structs.cc: Updated comment.
        * util.cc: Prevent segfault when the procedures_t(size_t)
        constructor is invoked with exactly 512 symbols.

libgcobol/ChangeLog:

        * charmaps.h (ascii_at): New define for '@',
        (charmap_load_unaligned): Eliminate.
        (charmap_store_unaligned): Eliminate.
        (store_uint16): Eliminate.
        (store_uint32): Eliminate.
        (class charmap_t): Make m_ascii_sign_bit, ::getch, ::putch, and
        ::memset behave for big-endian architectures.
        * gfileio.cc (write_a_char): Behave properly on big-endian
        architectures.
        (sequential_file_write): Likewise.
        (line_sequential_file_read_sbc): Likewise.
        (line_sequential_file_read): Likewise.
        * libgcobol.cc (edited_to_binary): Simplified.
        (int128_to_field): Proper big-endian behavior.
        (uber_compare): Proper big-endian behavior, simplified ABI.
        (__gg__dirty_to_binary): Simplified.
        (compare_strings): Use new uber_compare call.
        (__gg__compare_2): Likewise.
---
 gcc/cobol/genapi.cc    |  23 ++-
 gcc/cobol/structs.cc   |   2 +-
 gcc/cobol/util.cc      |   5 +-
 libgcobol/charmaps.h   | 320 +++++++++++++++++++++++++----------------
 libgcobol/gfileio.cc   |  39 +++--
 libgcobol/libgcobol.cc | 199 +++++++++++--------------
 6 files changed, 330 insertions(+), 258 deletions(-)

diff --git a/gcc/cobol/genapi.cc b/gcc/cobol/genapi.cc
index 5fc6c8d726e..33b5f09aaf1 100644
--- a/gcc/cobol/genapi.cc
+++ b/gcc/cobol/genapi.cc
@@ -461,6 +461,25 @@ get_level_88_domain(size_t parent_capacity,
cbl_field_t *var, size_t &returned_s
 
   // Numerics are converted to strings, and handled as above
 
+  /*  For example:
+
+       77 var-1 PIC 99V9.
+           88 var-1-z VALUE zero THRU 10.
+           88 var-1-big VALUE 20 THRU 40.
+           88 var-1-huge VALUE 40 THRU 999.
+           88 var-1-asc VALUE "U2" THRU "XYZZY".
+
+    Creates these four string segments:
+
+      "1FZ2A10"
+      "2A202A40"
+      "2A403A999"
+      "2AU25AXYZZY"
+
+    Each gets converted to UTF-32 as the initial value.
+
+    */
+
   size_t retval_capacity = 64;
   char *retval = static_cast<char *>(xmalloc(retval_capacity));
   size_t output_index = 0;
@@ -12754,7 +12773,9 @@ create_and_call(size_t narg,
       // These have to be passed to be passed by value.
       crv = by_value_e;
       }
-    else if( crv == by_value_e && args[i].refer.field->type ==
FldAlphanumeric)
+    else if(   crv == by_value_e
+            && args[i].refer.field->type == FldAlphanumeric
+            && (args[i].refer.field->attr & FIGCONST_MASK) !=
zero_value_e )
       {
       // Maybe passing an alphanumeric BY VALUE should be a syntax error?
       crv = by_content_e;
diff --git a/gcc/cobol/structs.cc b/gcc/cobol/structs.cc
index 37b2b0c878a..cacd6b4f73e 100644
--- a/gcc/cobol/structs.cc
+++ b/gcc/cobol/structs.cc
@@ -308,7 +308,7 @@ typedef struct cblc_file_t
     int                  errnum;           // most recent errno; can't
reuse "errno" as the name
     file_status_t        io_status;        // See 2014 standard, section
9.1.12
     int                  padding;          // Actually a char
-    uint32_t             delimiter;        // ends a record; defaults to
'\n'.
+    uint32_t             delimiter;        // This is four bytes in
encoding order.
     int                  stride;           // Width of a character
     int                  flags;            // cblc_file_flags_t
     uint32_t             recent_char;      // This is the most recent
char sent to the file
diff --git a/gcc/cobol/util.cc b/gcc/cobol/util.cc
index 4a44eb2ae51..c4927a9c459 100644
--- a/gcc/cobol/util.cc
+++ b/gcc/cobol/util.cc
@@ -2684,8 +2684,9 @@ namespace match_proc {
   public:
     procedures_t( size_t program ) {
       // find sections and paragraphs
-      for( symbol_elem_t *e = symbols_begin(program+1); e->program ==
program; e++ ) {
-        if( e->type == SymLabel ) {
+      for( symbol_elem_t *e = symbols_begin(program+1);
+          e < symbols_end() && e->program == program; e++ ) {
+          if( e->type == SymLabel ) {
           const auto& L = *cbl_label_of(e);
           auto isym = e - symbols_begin();
           switch(L.type) {
diff --git a/libgcobol/charmaps.h b/libgcobol/charmaps.h
index 3f212ac93bc..626ef385234 100644
--- a/libgcobol/charmaps.h
+++ b/libgcobol/charmaps.h
@@ -271,6 +271,7 @@ enum
 #define ascii_query            ((uint8_t)('?'))
 #define ascii_lbrace           ((uint8_t)('{'))
 #define ascii_rbrace           ((uint8_t)('}'))
+#define ascii_at               ((uint8_t)('@'))
 #define ascii_ff               ((uint8_t)('\f'))
 #define ascii_return           ((uint8_t)('\r'))
 #define ascii_newline          ((uint8_t)('\n'))
@@ -340,40 +341,6 @@ charmap_as_unsigned_chars(const char *p)
   return reinterpret_cast<const unsigned char *>(p);
   }
 
-template <typename T>
-static T
-charmap_load_unaligned(const void *p)
-  {
-  static_assert(std::is_trivially_copyable<T>::value,
-                "charmap_load_unaligned requires a trivially copyable
type");
-  T retval;
-  std::memcpy(&retval, p, sizeof(retval));
-  return retval;
-  }
-
-template <typename T>
-static void
-charmap_store_unaligned(void *p, T value)
-  {
-  static_assert(std::is_trivially_copyable<T>::value,
-                "charmap_store_unaligned requires a trivially copyable
type");
-  std::memcpy(p, &value, sizeof(value));
-  }
-
-static inline void
-store_uint16(unsigned char *p, uint16_t value)
-  {
-  // This routine is handling encoded characters, so the storage is
literal
-  memcpy(p, &value, 2);
-  }
-
-static inline void
-store_uint32(unsigned char *p, uint32_t value)
-  {
-  // This routine is handling encoded characters, so the storage is
literal
-  memcpy(p, &value, 4);
-  }
-
 class charmap_t;
 
 /*
@@ -470,26 +437,27 @@ class charmap_t
       sign_type_ebcdic,
       } m_numeric_sign_type;
 
+    // In numeric display with sign internal, this bit gets turned on in
either
+    // the leading or trailing digit to indicate the value is negative.
It
+    // is the single bit turned on for the `@` character.
+    uint8_t m_ascii_sign_bit[4];
+
     // This map retains the ASCII-to-encoded value in m_encoding, so that
     // iconv need be called but once for each ASCII value.
     std::unordered_map<cbl_char_t, cbl_char_t> m_map_of_encodings;
 
-    const unsigned char *
-    skip_bom(const unsigned char *p, size_t outlength) const
-      {
-      if( m_has_bom && outlength >= 2 * m_stride )
-        {
-        p += m_stride;
-        }
-      return p;
-      }
-
     cbl_char_t
     get_encoded_char(const void *base_, size_t location) const
       {
+      // The idea here is that we look into a stream of encoded
characters.
+      // Starting at base_+location, we pick up m_stride characters and
put
+      // them into the cbl_char_t (which is 32-bit unsigned integer) so
that
+      // retval is not dependent on endianness of either the host machine
or
+      // the target machine.
+
       const unsigned char *base = static_cast<const unsigned char
*>(base_);
       const unsigned char *p = base + location;
-      cbl_char_t retval = 0;
+      cbl_char_t retval;
 
       switch(m_stride)
         {
@@ -501,17 +469,31 @@ class charmap_t
 
         case 2:
           {
-          uint16_t c;
-          memcpy(&c, p, 2);
-          retval = c;
+          if(m_is_big_endian)
+            {
+            // The first byte is the high-order byte
+            retval = (p[0]<<8) + p[1];
+            }
+          else
+            {
+            // The first byte is the low-order byte
+            retval = (p[1]<<8) + p[0];
+            }
           break;
           }
 
         default:
           {
-          uint32_t c;
-          memcpy(&c, p, 4);
-          retval = c;
+          if(m_is_big_endian)
+            {
+            // The first byte is the high-order byte
+            retval = (p[0]<<24) + (p[1]<<16) + (p[2]<<8) + p[3];
+            }
+          else
+            {
+            // The first byte is the low-order byte
+            retval = (p[3]<<24) + (p[2]<<16) + (p[1]<<8) + p[0];
+            }
           break;
           }
         }
@@ -522,6 +504,9 @@ class charmap_t
     void
     put_encoded_char(cbl_char_t ch, void *base_, size_t location) const
       {
+      // This is the reverse of get encoded character.  The value in ch
is
+      // placed in memory
+
       unsigned char *base = static_cast<unsigned char *>(base_);
       unsigned char *p = base + location;
 
@@ -532,11 +517,39 @@ class charmap_t
           break;
 
         case 2:
-          store_uint16(p, static_cast<uint16_t>(ch));
+          {
+          if(m_is_big_endian)
+            {
+            // The first byte is the high-order byte
+            p[0] = ch>>8;
+            p[1] = ch;
+            }
+          else
+            {
+            // The first byte is the low-order byte
+            p[1] = ch>>8;
+            p[0] = ch;
+            }
           break;
+          }
 
         default:
-          store_uint32(p, ch);
+          if(m_is_big_endian)
+            {
+            // The first byte is the high-order byte
+            p[0] = ch>>24;
+            p[1] = ch>>16;
+            p[2] = ch>>8;
+            p[3] = ch;
+            }
+          else
+            {
+            // The first byte is the low-order byte
+            p[3] = ch>>24;
+            p[2] = ch>>16;
+            p[1] = ch>>8;
+            p[0] = ch;
+            }
           break;
         }
       }
@@ -671,12 +684,25 @@ class charmap_t
             &outbuf, &outbytesleft);
       outlength = sizeof(response_) - outbytesleft;
       m_is_like_utf8 = (outlength == 3);
+
+      if( !is_like_ebcdic() )
+        {
+        memset(m_ascii_sign_bit, 0x00, 4);
+        if( m_is_big_endian )
+          {
+          m_ascii_sign_bit[m_stride-1] = 0x40;
+          }
+        else
+          {
+          m_ascii_sign_bit[0] = 0x40;
+          }
+        }
       }
 
-    bool is_valid()      const { return m_is_valid     ; }
-    bool is_big_endian() const { return m_is_big_endian; }
-    bool has_bom()       const { return m_has_bom      ; }
-    uint8_t stride()     const { return m_stride       ; }
+    bool is_valid()             const { return m_is_valid       ; }
+    bool is_big_endian()        const { return m_is_big_endian  ; }
+    bool has_bom()              const { return m_has_bom        ; }
+    uint8_t stride()            const { return m_stride         ; }
 
     cbl_char_t
     mapped_character(unsigned char ch)
@@ -696,45 +722,12 @@ class charmap_t
         {
         retval = 0;
         size_t outlength = 0;
-        char *mapped = __gg__iconverter(DEFAULT_SOURCE_ENCODING,
-                                        m_encoding,
-                                        &ch,
-                                        1,
-                                        &outlength);
-        size_t data_length = outlength;
-        const unsigned char *p = charmap_as_unsigned_chars(mapped);
-        if( m_has_bom && data_length >= 2 * stride() )
-          {
-          p = skip_bom(p, data_length);
-          data_length -= stride();
-          }
-
-        switch(stride())
-          {
-          case 1:
-            {
-            uint8_t c;
-            memcpy(&c, mapped, m_stride);
-            retval = c;
-            break;
-            }
-
-          case 2:
-            {
-            uint16_t c;
-            memcpy(&c, mapped, m_stride);
-            retval = c;
-            break;
-            }
-
-          case 4:
-            {
-            uint32_t c;
-            memcpy(&c, mapped, m_stride);
-            retval = c;
-            break;
-            }
-          }
+        const char *mapped = __gg__iconverter(DEFAULT_SOURCE_ENCODING,
+                                              m_encoding,
+                                              &ch,
+                                              1,
+                                              &outlength);
+        retval = get_encoded_char(mapped, 0);
         m_map_of_encodings[ch] = retval;
         }
       return retval;
@@ -821,7 +814,7 @@ class charmap_t
     switch(m_numeric_sign_type)
       {
       case sign_type_ascii:
-        retval = !!(digit & NUMERIC_DISPLAY_SIGN_BIT_ASCII);
+        retval = !!(digit & m_ascii_sign_bit[m_stride-1]);
         break;
 
       case sign_type_ebcdic:
@@ -838,34 +831,31 @@ class charmap_t
     // ebcdic.
     switch(m_numeric_sign_type)
       {
+      // We need to do this in a loop because of the headaches caused by
+      // dealing with, for instance, little-endian characters on a
big-endian
+      // architecture.
       case sign_type_ascii:
         {
-        uint32_t the_bit = m_is_big_endian
-                         ? NUMERIC_DISPLAY_SIGN_BIT_ASCII << (m_stride-1)
* 8
-                         : NUMERIC_DISPLAY_SIGN_BIT_ASCII;
         if( is_negative )
           {
-          digit |= the_bit;
+          digit |= m_ascii_sign_bit[m_stride-1];
           }
         else
           {
-          digit &= ~the_bit;
+          digit &= ~m_ascii_sign_bit[m_stride-1];
           }
         break;
         }
 
       case sign_type_ebcdic:
         {
-        uint32_t the_bit = m_is_big_endian
-                         ? NUMERIC_DISPLAY_SIGN_BIT_EBCDIC <<
(m_stride-1) * 8
-                         : NUMERIC_DISPLAY_SIGN_BIT_EBCDIC;
         if( is_negative )
           {
-          digit &= ~the_bit;
+          digit &= ~NUMERIC_DISPLAY_SIGN_BIT_EBCDIC;
           }
         else
           {
-          digit |= the_bit;
+          digit |= NUMERIC_DISPLAY_SIGN_BIT_EBCDIC;
           }
         break;
         }
@@ -873,6 +863,52 @@ class charmap_t
     return digit;
     }
 
+  void
+  set_streamed_digit_negative(uint8_t *digit, bool is_negative)
+    {
+    // Enter with digit pointing to a digit that needs to be adjusted for
+    // numeric-display internal signededness.
+
+    // The loop might look odd, but it's how I decided to handle issues
of
+    // big-endian characters on little-endian architectures, and
+    // little-endian characters on big-endian architectures, and so on.
+    switch(m_numeric_sign_type)
+      {
+      case sign_type_ascii:
+        {
+        if( is_negative )
+          {
+          for(int i=0; i<m_stride; i++ )
+            {
+            digit[i] |= m_ascii_sign_bit[i];
+            }
+          }
+        else
+          {
+          for(int i=0; i<m_stride; i++ )
+            {
+            digit[i] &= ~m_ascii_sign_bit[i];
+            }
+          }
+        break;
+        }
+
+      case sign_type_ebcdic:
+        {
+        if( is_negative )
+          {
+          *digit &= ~NUMERIC_DISPLAY_SIGN_BIT_EBCDIC;
+          }
+        else
+          {
+          *digit |= NUMERIC_DISPLAY_SIGN_BIT_EBCDIC;
+          }
+        break;
+        }
+      }
+    }
+
+
   bool
   is_like_ebcdic() const
     {
@@ -888,6 +924,10 @@ class charmap_t
   void
   memset(void *dest_, cbl_char_t ch, size_t bytelength)
     {
+    uint8_t byte3 = ch >> 24;
+    uint8_t byte2 = ch >> 16;
+    uint8_t byte1 = ch >>  8;
+    uint8_t byte0 = ch      ;
     unsigned char *dest = static_cast<unsigned char *>(dest_);
     switch(m_stride)
       {
@@ -902,10 +942,6 @@ class charmap_t
           {
           // We are being asked to fill a byte-wide buffer with a
multi-byte
           // character.
-          unsigned char byte3 = static_cast<unsigned char>(ch >> 24);
-          unsigned char byte2 = static_cast<unsigned char>(ch >> 16);
-          unsigned char byte1 = static_cast<unsigned char>(ch >>  8);
-          unsigned char byte0 = static_cast<unsigned char>(ch);
           size_t i = 0;
           if( byte3 )
             {
@@ -949,31 +985,56 @@ class charmap_t
         // We know the target has an even number of bytes available.  We
also
         // know that each codepoint is usually one, but sometimes two,
pairs
         // of bytes.
-        uint16_t top_half = static_cast<uint16_t>(ch >> 16);
-        uint16_t bottom_half = static_cast<uint16_t>(ch);
         size_t i = 0;
         while( i < bytelength )
           {
-          if( top_half )
+          if( byte3 | byte2 )
             {
             if( i + 4 <= bytelength )
               {
-              store_uint16(dest + i, top_half);
-              i += 2;
-              store_uint16(dest + i, bottom_half);
-              i += 2;
+              if( m_is_big_endian )
+                {
+                dest[i+0] = byte3;
+                dest[i+1] = byte2;
+                dest[i+2] = byte1;
+                dest[i+3] = byte0;
+                }
+              else
+                {
+                dest[i+3] = byte3;
+                dest[i+2] = byte2;
+                dest[i+1] = byte1;
+                dest[i+0] = byte0;
+                }
+              i += 4;
               }
             else
               {
-              store_uint16(dest + i,
-                           static_cast<uint16_t>(
-
mapped_character(ascii_space)));
+              if( m_is_big_endian)
+                {
+                dest[i+1] = ascii_space;
+                dest[i+0] = 0;
+                }
+              else
+                {
+                dest[i+1] = 0;
+                dest[i+0] = ascii_space;
+                }
               i += 2;
               }
             }
           else
             {
-            store_uint16(dest + i, bottom_half);
+            if( m_is_big_endian )
+              {
+              dest[i+0] = byte1;
+              dest[i+1] = byte0;
+              }
+            else
+              {
+              dest[i+1] = byte1;
+              dest[i+0] = byte0;
+              }
             i += 2;
             }
           }
@@ -986,7 +1047,20 @@ class charmap_t
         // We know the target has a multiple of four bytes available.
         for( size_t i = 0; i < bytelength; i += 4 )
           {
-          store_uint32(dest + i, ch);
+          if( m_is_big_endian )
+            {
+            dest[i+0] = byte3;
+            dest[i+1] = byte2;
+            dest[i+2] = byte1;
+            dest[i+3] = byte0;
+            }
+          else
+            {
+            dest[i+3] = byte3;
+            dest[i+2] = byte2;
+            dest[i+1] = byte1;
+            dest[i+0] = byte0;
+            }
           }
         break;
         }
diff --git a/libgcobol/gfileio.cc b/libgcobol/gfileio.cc
index dba7001237b..934541e08ea 100644
--- a/libgcobol/gfileio.cc
+++ b/libgcobol/gfileio.cc
@@ -360,7 +360,7 @@ __gg__file_init(
   {
   if( !(file->flags & file_flag_initialized_e) )
     {
-    charmap_t *charmap = __gg__get_charmap(encoding);
+    const charmap_t *charmap = __gg__get_charmap(encoding);
 
     file->name                = strdup(name);
     file->symbol_table_index  = symbol_table_index;
@@ -386,7 +386,6 @@ __gg__file_init(
     file->access              = (cbl_file_access_t)access ;
     file->errnum              = 0 ;
     file->io_status           = FsSuccess ;
-    file->delimiter           = charmap->mapped_character(ascii_newline)
;
     file->stride              = charmap->stride();
     file->flags               = file_flag_none_e;
         file->flags          |= (optional ? file_flag_optional_e :
file_flag_none_e)
@@ -398,6 +397,18 @@ __gg__file_init(
     file->encoding            = encoding;
     file->alphabet            = alphabet;
 
+    // Note: eventually the delimiter needs to be a variable; it can be
set
+    // by the programmer.
+    size_t nbytes;
+    const char ch = ascii_newline;
+    const char *delim = __gg__iconverter(DEFAULT_SOURCE_ENCODING,
+                                         file->encoding,
+                                         &ch,
+                                         1,
+                                         &nbytes);
+    memset(&file->delimiter, 0, 4);
+    memcpy(&file->delimiter, delim, file->stride);
+
     if( file->access == file_inaccessible_e )
       {
       file->access = file_access_seq_e;
@@ -2669,9 +2680,11 @@ static void
 write_a_char(cblc_file_t *file, cbl_char_t ch)
   {
   size_t nbytes;
+  // Whether big- or little-endian, this will give us the character we
want
+  unsigned char uch = ch % 256;
   const char *converted = __gg__iconverter(DEFAULT_SOURCE_ENCODING,
                                            file->encoding,
-                                           &ch,
+                                           &uch,
                                            1,
                                            &nbytes);
   fwrite(converted, nbytes, 1, file->file_pointer);
@@ -2730,7 +2743,7 @@ sequential_file_write(cblc_file_t    *file,
 
   if( file->org == file_line_sequential_e )
     {
-    // If file-sequential, then trailing spaces are removed:
+    // If line-sequential, then trailing spaces are removed:
     while(bytes_to_write > 0
            && charmap->getch(location, bytes_to_write-stride)
                                   ==
charmap->mapped_character(ascii_space) )
@@ -3187,8 +3200,9 @@ line_sequential_file_read_sbc(cblc_file_t *file,
char space)
       }
     // Much hinges on where the next newline is to be found:
     pstart = file->buffer+file->buffer_pos;
+    char ch = reinterpret_cast<const char *>(&file->delimiter)[0];
     pnewline = reinterpret_cast<const char *>(memchr(pstart,
-                      static_cast<char>(file->delimiter),
+                      ch,
                       file->buffer_len - file->buffer_pos));
     if( file->buffer_pos >= file->buffer_len )
       {
@@ -3313,8 +3327,9 @@ line_sequential_file_read_sbc(cblc_file_t *file,
char space)
           }
         }
       pstart = file->buffer+file->buffer_pos;
+      char ch = reinterpret_cast<const char *>(&file->delimiter)[0];
       pnewline = reinterpret_cast<const char *>(memchr(pstart,
-                        static_cast<char>(file->delimiter),
+                        ch,
                         file->buffer_len - file->buffer_pos));
       if( pnewline )
         {
@@ -3380,7 +3395,7 @@ line_sequential_file_read(  cblc_file_t *file)
   // it makes more sense to me.
 
   // We first stage the data into the record area.
-  cbl_char_t ch;
+  uint8_t ch[4];
 
   long fpos = static_cast<long>(file->file_fpos);
 
@@ -3427,11 +3442,10 @@ line_sequential_file_read(  cblc_file_t *file)
 
     // There are still characters in the file->buffer, and we are still
looking
     // to fill the record_area, and we are still looking for a
end-of-line.
-    ch = 0;
-    memcpy(&ch, file->buffer+file->buffer_pos, stride);
+    memcpy(ch, file->buffer+file->buffer_pos, stride);
     file->buffer_pos += stride;
     file->file_fpos += stride;
-    if( ch == file->delimiter )
+    if( memcmp(ch, &file->delimiter, stride) == 0)
       {
       break;
       }
@@ -3485,13 +3499,12 @@ line_sequential_file_read(  cblc_file_t *file)
           goto done;
           }
         }
-      ch = 0;
-      memcpy(&ch, file->buffer+file->buffer_pos, stride);
+      memcpy(ch, file->buffer+file->buffer_pos, stride);
       file->buffer_pos += stride;
       file->file_fpos += stride;
       // We can't use handle_ferror() directly, because an EOF is
       // a legitimate way to end the last line.
-      if( ch == file->delimiter )
+    if( memcmp(ch, &file->delimiter, stride) == 0 )
         {
         clearerr(file->file_pointer);
         break;
diff --git a/libgcobol/libgcobol.cc b/libgcobol/libgcobol.cc
index cdc65d6b488..52e3d929c75 100644
--- a/libgcobol/libgcobol.cc
+++ b/libgcobol/libgcobol.cc
@@ -679,13 +679,10 @@ __gg__decimal_point_is_comma()
 
 static __int128
 edited_to_binary( const cblc_field_t *field,
-                  char *ps_,
-                  size_t length,
-                  int *rdigits)
+                  const char         *ps_,
+                        size_t        length,
+                        int          *rdigits)
   {
-  charmap_t *charmap = __gg__get_charmap(field->encoding);
-
-  const unsigned char *ps = as_unsigned_chars(ps_);
   // This routine is used for converting NumericEdited strings to
   // binary.
 
@@ -697,7 +694,13 @@ edited_to_binary( const cblc_field_t *field,
   // result as negative.  We are going to look for a decimal point and
count up
   // the numerical digits to the right of it.  And we are going to
pretend
   // that nothing else matters.
-
+  size_t nbytes;
+  const unsigned char *ps = reinterpret_cast<unsigned char *>
+
(__gg__iconverter(field->encoding,
+
DEFAULT_SOURCE_ENCODING,
+                                                      ps_,
+                                                      length,
+                                                      &nbytes));
   int hyphen = 0;
   *rdigits = 0;
 
@@ -716,37 +719,28 @@ edited_to_binary( const cblc_field_t *field,
 
   while( index < length )
     {
-    cbl_char_t ch = charmap->getch(ps, &index);
+    unsigned char ch = ps[index++];
 
     // Save the last two characters for the DB/CR test:
     chm2 = chm1;
     chm1 = ch;
 
-    if( ch == charmap->mapped_character(__gg__decimal_point) )
+    if( ch == __gg__decimal_point )
       {
       delta_r = 1;
       continue;
       }
-    if( ch == charmap->mapped_character(ascii_minus)  )
+    if( ch == ascii_minus  )
       {
       hyphen = 1;
       continue;
       }
 
-    if(  ch >= charmap->mapped_character(ascii_0)
-      && ch <= charmap->mapped_character(ascii_9) )
+    if(  ch >= ascii_0
+      && ch <= ascii_9 )
       {
-      uint8_t the_byte;
-      if( charmap->is_big_endian() )
-        {
-        the_byte = ch >> ((charmap->stride()-1)*8);
-        }
-      else
-        {
-        the_byte = ch;
-        }
       result *= 10;
-      result += the_byte & 0x0F ;
+      result += ch & 0x0F ;
       *rdigits += delta_r ;
       continue;
       }
@@ -756,17 +750,17 @@ edited_to_binary( const cblc_field_t *field,
   // is negative:
   if( !hyphen && length >= 2)
     {
-    if(        (   chm2 == charmap->mapped_character(ascii_D)
-                || chm2 == charmap->mapped_character(ascii_d))
-            && (   chm1 == charmap->mapped_character(ascii_B)
-                || chm1 == charmap->mapped_character(ascii_b)) )
+    if(        (   chm2 == ascii_D
+                || chm2 == ascii_d)
+            && (   chm1 == ascii_B
+                || chm1 == ascii_b) )
       {
       hyphen = 1;
       }
-    else if(   (   chm2 == charmap->mapped_character(ascii_C)
-                || chm2 == charmap->mapped_character(ascii_c))
-            && (   chm1 == charmap->mapped_character(ascii_R)
-                || chm1 == charmap->mapped_character(ascii_r)) )
+    else if(   (   chm2 == ascii_C
+                || chm2 == ascii_c)
+            && (   chm1 == ascii_R
+                || chm1 == ascii_r) )
       {
       hyphen = 1;
       }
@@ -2245,10 +2239,10 @@ int128_to_field(cblc_field_t   *var,
 
                 // First, convert the binary value to the correct-length
string
                 size_error =
-                  __gg__binary_to_string_encoded(as_chars( location),
-                                                  var->digits,
-                                                  value,
-                                                  var->encoding);
+                          __gg__binary_to_string_encoded(as_chars(
location),
+                                                          var->digits,
+                                                          value,
+                                                          var->encoding);
 
                 // Check for a size error on a negative value.  It
conceivably
                 // was truncated down to zero, in which case we need to
@@ -2256,8 +2250,7 @@ int128_to_field(cblc_field_t   *var,
                 if( size_error && is_negative )
                   {
                   // If all of the digits are zero, then the result is
zero,
-                  // and
-                  // we have to kill the is_negative flag:
+                  // and we have to kill the is_negative flag:
                   is_negative = false;
                   size_t index = 0;
                   while(index<length)
@@ -2272,13 +2265,10 @@ int128_to_field(cblc_field_t   *var,
                   }
 
                 unsigned char *sign_location =
-                  var->attr & leading_e ? location
-                                        : location + length - stride;
-                cbl_char_t sign_digit = charmap->getch(sign_location,
-                                                       (size_t)0);
-                sign_digit = charmap->set_digit_negative(sign_digit,
-                                                         is_negative);
-                charmap->putch(sign_digit, sign_location, (size_t)0);
+                            var->attr & leading_e ? location
+                                                  : location + length -
stride;
+                charmap->set_streamed_digit_negative(sign_location,
+                                                     is_negative);
                 }
               }
             else
@@ -2739,24 +2729,13 @@ uint32_t collation_position( cbl_char_t ch )
 
 static cbl_char_t
 uber_compare(cbl_char_t ch_left,
-             cbl_char_t ch_right,
-             const charmap_t *charmap)
+             cbl_char_t ch_right)
   {
-  if( charmap->is_big_endian() )
-    {
-    // This simple hack works when the big-endian characters nonetheless
fit
-    // into zero through 255.  This is short-sighted.  I need glasses.
Again,
-    // I do not know how to fix this until multi-byte collation is
addressed.
-    ch_left  >>= 8*(charmap->stride()-1);
-    ch_right >>= 8*(charmap->stride()-1);
-    }
-
+  // This is where collation is going to have to be fixed for multi-byte
+  // encodings.  For now, if both characters fit into 0xFF, then we will
+  // use the current collation.  Otherwise, we just compare them.
   if( ((ch_left | ch_right) & 0xFFFFFF00) == 0x00000000 )
     {
-    // This is where collation is going to have to be fixed for
multi-byte
-    // encodings.  For now, if both characters fit into 0xFF, then we
will
-    // use the current collation.  Otherwise, we just compare them
-
     // Both characters fit into the current DISPLAY codeset, so assume we
     // are using the DISPLAY collation:
     ch_left  = collated(ch_left);
@@ -2770,7 +2749,6 @@ uber_compare(cbl_char_t ch_left,
   return retval;
   }
 
-
 extern "C"
 int
 __gg__setop_compare(
@@ -2945,9 +2923,9 @@ __gg__dirty_to_binary_source(const char *dirty,
 
 extern "C"
 __int128
-__gg__dirty_to_binary(const char *dirty,
+__gg__dirty_to_binary(const char *dirty_in,
                       cbl_encoding_t encoding,
-                      int length,
+                      int length_in,
                       int *rdigits)
   {
   // This routine is used for converting uncontrolled strings to a
@@ -2969,17 +2947,12 @@ __gg__dirty_to_binary(const char *dirty,
   // We are limiting the number of digits in the number to
   // MAX_FIXED_POINT_DIGITS
 
-  charmap_t *charmap    = __gg__get_charmap(encoding);
-  int stride = charmap->stride();
-
-  cbl_char_t mapped_minus          =
charmap->mapped_character(ascii_minus);
-  cbl_char_t mapped_plus           =
charmap->mapped_character(ascii_plus);
-  cbl_char_t mapped_decimal_point =
-    charmap->mapped_character(__gg__decimal_point);
-  cbl_char_t mapped_0              = charmap->mapped_character(ascii_0);
-  cbl_char_t mapped_9              = charmap->mapped_character(ascii_9);
-  cbl_char_t mapped_E              = charmap->mapped_character(ascii_E);
-  cbl_char_t mapped_e              = charmap->mapped_character(ascii_e);
+  size_t length;
+  const char *dirty = __gg__iconverter(encoding,
+                                       DEFAULT_SOURCE_ENCODING,
+                                       dirty_in,
+                                       length_in,
+                                       &length);
 
   __int128 retval = 0;
 
@@ -2993,37 +2966,33 @@ __gg__dirty_to_binary(const char *dirty,
   int delta_r = 0;
 
   // We now loop over the remaining input characters:
-  cbl_char_t ch = '\0';
+  char ch = '\0';
   size_t chindex = 0;
 
   if(length > 0)
     {
-    length -= stride;
-    ch = charmap->getch(dirty, &chindex);
-    if( ch == mapped_minus )
+    length -= 1;
+    ch = dirty[chindex++];
+    if( ch == ascii_minus )
       {
       hyphen = 1;
       }
-    else if( ch == mapped_plus )
+    else if( ch == ascii_plus )
       {
       // A plus sign is okay
       }
-    else if( ch == mapped_decimal_point )
+    else if( ch == __gg__decimal_point )
       {
       delta_r = 1;
       }
-    else if( ch >= mapped_0
-          && ch <= mapped_9  )
+    else if( ch >= ascii_0
+          && ch <= ascii_9  )
       {
-      retval = ch - mapped_0 ;
+      retval = ch - ascii_0 ;
       if( retval )
         {
         digit_count += 1;
         }
-      if( charmap->is_big_endian() )
-        {
-        retval >>= 8*(stride-1);
-        }
       }
     else
       {
@@ -3036,17 +3005,16 @@ __gg__dirty_to_binary(const char *dirty,
 
   while( length > 0 )
     {
-    length -= stride;
-    ch = charmap->getch(dirty, &chindex);
-    if( ch == mapped_decimal_point && delta_r == 0 )
+    ch = dirty[chindex++];
+    if( ch == __gg__decimal_point && delta_r == 0 )
       {
       // This is the first decimal point we've seen, so we
       // can start counting rdigits:
       delta_r = 1;
       continue;
       }
-    if(    ch < mapped_0
-        || ch > mapped_9 )
+    if(    ch < ascii_0
+        || ch > ascii_9 )
       {
       // When we hit something that isn't a digit, then we are done
       break;
@@ -3054,12 +3022,7 @@ __gg__dirty_to_binary(const char *dirty,
     if( digit_count < MAX_FIXED_POINT_DIGITS )
       {
       retval *= 10;
-      ch -= mapped_0;
-      if( charmap->is_big_endian() )
-        {
-        ch >>= 8*(stride-1);
-        }
-
+      ch -= ascii_0;
       retval += ch ;
       *rdigits += delta_r;
       if( retval )
@@ -3070,38 +3033,38 @@ __gg__dirty_to_binary(const char *dirty,
     }
 
   // Let's check for an exponent:
-  if(   ch == mapped_E
-     || ch == mapped_e )
+  if(   ch == ascii_E
+     || ch == ascii_e )
     {
     int exponent = 0;
     int exponent_sign = 1;
     if( length > 0  )
       {
-      ch = charmap->getch(dirty, chindex);
-      if( ch == mapped_plus)
+      ch = dirty[chindex];
+      if( ch == ascii_plus)
         {
-        length -= stride;
-        dirty += stride;
+        length -= 1;
+        dirty += 1;
         }
-      else if( ch == mapped_minus )
+      else if( ch == ascii_minus )
         {
         exponent_sign = -1;
-        length -= stride;
-        dirty += stride;
+        length -= 1;
+        dirty += 1;
         }
       }
     while(length > 0)
       {
-      length -= stride;
-      ch = charmap->getch(dirty, &chindex);
-      if(    ch < mapped_0
-          || ch > mapped_9 )
+      length -= 1;
+      ch = dirty[chindex++];
+      if(    ch < ascii_0
+          || ch > ascii_9 )
         {
         // When we hit something that isn't a digit, then we are done
         break;
         }
       exponent *= 10;
-      exponent += ch - mapped_0 ;
+      exponent += ch - ascii_0 ;
       }
     exponent *= exponent_sign;
     // We need to adjust the retval and the rdigits based on the
exponent.
@@ -5054,7 +5017,7 @@ compare_strings(char   *left_string,
     {
     cbl_char_t ch_left  = charmap_left->getch(left_string, &index_left);
     cbl_char_t ch_right = charmap_right->getch(right_string,
&index_right);
-    retval = uber_compare(ch_left, ch_right, charmap_left);
+    retval = uber_compare(ch_left, ch_right);
     }
 
   // We need to space-extend the shorter value.  That's because
@@ -5065,7 +5028,7 @@ compare_strings(char   *left_string,
       {
       cbl_char_t ch_left  = charmap_left->getch(left_string,
&index_left);
       cbl_char_t ch_right = charmap_right->mapped_character(ascii_space);
-      retval = uber_compare(ch_left, ch_right, charmap_left);
+      retval = uber_compare(ch_left, ch_right);
       }
     }
   else
@@ -5077,7 +5040,7 @@ compare_strings(char   *left_string,
       index_right %= right_length;
       cbl_char_t ch_left  = charmap_left->getch(left_string,
&index_left);
       cbl_char_t ch_right = charmap_right->getch(right_string,
&index_right);
-      retval = uber_compare(ch_left, ch_right, charmap_left);
+      retval = uber_compare(ch_left, ch_right);
       }
     }
 
@@ -5087,7 +5050,7 @@ compare_strings(char   *left_string,
       {
       cbl_char_t ch_left  = charmap_left->mapped_character(ascii_space);
       cbl_char_t ch_right = charmap_right->getch(right_string,
&index_right);
-      retval = uber_compare(ch_left, ch_right, charmap_left);
+      retval = uber_compare(ch_left, ch_right);
       }
     }
   else
@@ -5097,7 +5060,7 @@ compare_strings(char   *left_string,
       index_left %= left_length;
       cbl_char_t ch_left  = charmap_left->mapped_character(ascii_space);
       cbl_char_t ch_right = charmap_right->getch(right_string,
&index_right);
-      retval = uber_compare(ch_left, ch_right, charmap_left);
+      retval = uber_compare(ch_left, ch_right);
       }
     }
 
@@ -5174,7 +5137,7 @@ __gg__compare_2(cblc_field_t  *left_side,
   if( left_figconst && right_figconst )
     {
     // We are comparing two figurative constants
-    retval = uber_compare(fig_left, fig_right, charmap_left);
+    retval = uber_compare(fig_left, fig_right);
     compare = true;
     goto fixup_retval;
     }
@@ -5213,7 +5176,7 @@ __gg__compare_2(cblc_field_t  *left_side,
           cbl_char_t fig_of_right =
 
charmap_left->figconst_character(right_figconst);
           cbl_char_t left_ch = charmap_left->getch(left_location, i);
-          retval = uber_compare(left_ch, fig_of_right, charmap_left);
+          retval = uber_compare(left_ch, fig_of_right);
           if( retval )
             {
             break;
@@ -5271,7 +5234,7 @@ __gg__compare_2(cblc_field_t  *left_side,
               unsigned int fig_of_right =
 
charmap_left->figconst_character(right_figconst);
               cbl_char_t ch_left = charmap_left->getch(left_location, i);
-              retval = uber_compare(ch_left, fig_of_right, charmap_left);
+              retval = uber_compare(ch_left, fig_of_right);
               if( retval )
                 {
                 break;
-- 
2.34.1

Reply via email to