LCOV - code coverage report
Current view: top level - port - cpl_recode_stub.cpp (source / functions) Hit Total Coverage
Test: gdal_filtered.info Lines: 206 312 66.0 %
Date: 2026-09-30 13:58:12 Functions: 11 11 100.0 %

          Line data    Source code
       1             : /**********************************************************************
       2             :  *
       3             :  * Name:     cpl_recode_stub.cpp
       4             :  * Project:  CPL - Common Portability Library
       5             :  * Purpose:  Character set recoding and char/wchar_t conversions, stub
       6             :  *           implementation to be used if iconv() functionality is not
       7             :  *           available.
       8             :  * Author:   Frank Warmerdam, warmerdam@pobox.com
       9             :  *
      10             :  * The bulk of this code is derived from the utf.c module from FLTK. It
      11             :  * was originally downloaded from:
      12             :  *    http://svn.easysw.com/public/fltk/fltk/trunk/src/utf.c
      13             :  *
      14             :  **********************************************************************
      15             :  * Copyright (c) 2008, Frank Warmerdam
      16             :  * Copyright 2006 by Bill Spitzak and others.
      17             :  * Copyright (c) 2009-2014, Even Rouault <even dot rouault at spatialys.com>
      18             :  *
      19             :  * Permission to use, copy, modify, and distribute this software for any
      20             :  * purpose with or without fee is hereby granted, provided that the above
      21             :  * copyright notice and this permission notice appear in all copies.
      22             :  *
      23             :  * THE SOFTWARE IS PROVIDED "AS IS" AND THE AUTHOR DISCLAIMS ALL WARRANTIES
      24             :  * WITH REGARD TO THIS SOFTWARE INCLUDING ALL IMPLIED WARRANTIES OF
      25             :  * MERCHANTABILITY AND FITNESS. IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR
      26             :  * ANY SPECIAL, DIRECT, INDIRECT, OR CONSEQUENTIAL DAMAGES OR ANY DAMAGES
      27             :  * WHATSOEVER RESULTING FROM LOSS OF USE, DATA OR PROFITS, WHETHER IN AN
      28             :  * ACTION OF CONTRACT, NEGLIGENCE OR OTHER TORTIOUS ACTION, ARISING OUT OF
      29             :  * OR IN CONNECTION WITH THE USE OR PERFORMANCE OF THIS SOFTWARE.
      30             :  **********************************************************************/
      31             : 
      32             : #include "cpl_port.h"
      33             : #include "cpl_string.h"
      34             : 
      35             : #include <cstring>
      36             : 
      37             : #include "cpl_conv.h"
      38             : #include "cpl_error.h"
      39             : #include "cpl_character_sets.c"
      40             : 
      41             : static unsigned utf8decode(const char *p, const char *end, int *len);
      42             : static unsigned utf8towc(const char *src, unsigned srclen, wchar_t *dst,
      43             :                          unsigned dstlen);
      44             : static unsigned utf8toa(const char *src, unsigned srclen, char *dst,
      45             :                         unsigned dstlen);
      46             : static unsigned utf8fromwc(char *dst, unsigned dstlen, const wchar_t *src,
      47             :                            unsigned srclen);
      48             : static unsigned utf8froma(char *dst, unsigned dstlen, const char *src,
      49             :                           unsigned srclen);
      50             : static int utf8test(const char *src, unsigned srclen);
      51             : 
      52             : #ifdef _WIN32
      53             : 
      54             : #include <windows.h>
      55             : #include <winnls.h>
      56             : 
      57             : static char *CPLWin32Recode(const char *src, unsigned src_code_page,
      58             :                             unsigned dst_code_page) CPL_RETURNS_NONNULL;
      59             : #endif
      60             : 
      61             : /* used by cpl_recode.cpp */
      62             : extern void CPLClearRecodeStubWarningFlags();
      63             : extern char *CPLRecodeStub(const char *, const char *,
      64             :                            const char *) CPL_RETURNS_NONNULL;
      65             : extern char *CPLRecodeFromWCharStub(const wchar_t *, const char *,
      66             :                                     const char *);
      67             : extern wchar_t *CPLRecodeToWCharStub(const char *, const char *, const char *);
      68             : 
      69             : /************************************************************************/
      70             : /* ==================================================================== */
      71             : /*      Stub Implementation not depending on iconv() or WIN32 API.      */
      72             : /* ==================================================================== */
      73             : /************************************************************************/
      74             : 
      75             : static bool bHaveWarned1 = false;
      76             : static bool bHaveWarned2 = false;
      77             : static bool bHaveWarned3 = false;
      78             : static bool bHaveWarned4 = false;
      79             : #ifdef _WIN32
      80             : static bool bHaveWarned5 = false;
      81             : static bool bHaveWarned6 = false;
      82             : #endif
      83             : 
      84             : /************************************************************************/
      85             : /*                   CPLClearRecodeStubWarningFlags()                   */
      86             : /************************************************************************/
      87             : 
      88       13461 : void CPLClearRecodeStubWarningFlags()
      89             : {
      90       13461 :     bHaveWarned1 = false;
      91       13461 :     bHaveWarned2 = false;
      92       13461 :     bHaveWarned3 = false;
      93       13461 :     bHaveWarned4 = false;
      94             : #ifdef _WIN32
      95             :     bHaveWarned5 = false;
      96             :     bHaveWarned6 = false;
      97             : #endif
      98       13461 : }
      99             : 
     100             : /************************************************************************/
     101             : /*                           CPLRecodeStub()                            */
     102             : /************************************************************************/
     103             : 
     104             : /**
     105             :  * Convert a string from a source encoding to a destination encoding.
     106             :  *
     107             :  * The only guaranteed supported encodings are CPL_ENC_UTF8, CPL_ENC_ASCII
     108             :  * and CPL_ENC_ISO8859_1. Currently, the following conversions are supported :
     109             :  * <ul>
     110             :  *  <li>CPL_ENC_ASCII -> CPL_ENC_UTF8 or CPL_ENC_ISO8859_1 (no conversion in
     111             :  *  fact)</li>
     112             :  *  <li>CPL_ENC_ISO8859_1 -> CPL_ENC_UTF8</li>
     113             :  *  <li>CPL_ENC_UTF8 -> CPL_ENC_ISO8859_1</li>
     114             :  * </ul>
     115             :  *
     116             :  * If an error occurs an error may, or may not be posted with CPLError().
     117             :  *
     118             :  * @param pszSource a NULL terminated string.
     119             :  * @param pszSrcEncoding the source encoding.
     120             :  * @param pszDstEncoding the destination encoding.
     121             :  *
     122             :  * @return a NULL terminated string which should be freed with CPLFree().
     123             :  */
     124             : 
     125     2035160 : char *CPLRecodeStub(const char *pszSource, const char *pszSrcEncoding,
     126             :                     const char *pszDstEncoding)
     127             : 
     128             : {
     129             :     /* -------------------------------------------------------------------- */
     130             :     /*      If the source or destination is current locale(), we change     */
     131             :     /*      it to ISO8859-1 since our stub implementation does not          */
     132             :     /*      attempt to address locales properly.                            */
     133             :     /* -------------------------------------------------------------------- */
     134             : 
     135     2035160 :     if (pszSrcEncoding[0] == '\0')
     136           0 :         pszSrcEncoding = CPL_ENC_ISO8859_1;
     137             : 
     138     2035160 :     if (pszDstEncoding[0] == '\0')
     139           0 :         pszDstEncoding = CPL_ENC_ISO8859_1;
     140             : 
     141             :     /* -------------------------------------------------------------------- */
     142             :     /*      ISO8859 to UTF8                                                 */
     143             :     /* -------------------------------------------------------------------- */
     144     2035160 :     if (strcmp(pszSrcEncoding, CPL_ENC_ISO8859_1) == 0 &&
     145     1958380 :         strcmp(pszDstEncoding, CPL_ENC_UTF8) == 0)
     146             :     {
     147     1958380 :         const int nCharCount = static_cast<int>(strlen(pszSource));
     148     1958380 :         char *pszResult = static_cast<char *>(CPLCalloc(1, nCharCount * 2 + 1));
     149             : 
     150     1958380 :         utf8froma(pszResult, nCharCount * 2 + 1, pszSource, nCharCount);
     151             : 
     152     1958380 :         return pszResult;
     153             :     }
     154             : 
     155             :     /* -------------------------------------------------------------------- */
     156             :     /*      UTF8 to ISO8859                                                 */
     157             :     /* -------------------------------------------------------------------- */
     158       76774 :     if (strcmp(pszSrcEncoding, CPL_ENC_UTF8) == 0 &&
     159       49447 :         strcmp(pszDstEncoding, CPL_ENC_ISO8859_1) == 0)
     160             :     {
     161       49447 :         int nCharCount = static_cast<int>(strlen(pszSource));
     162       49447 :         char *pszResult = static_cast<char *>(CPLCalloc(1, nCharCount + 1));
     163             : 
     164       49447 :         utf8toa(pszSource, nCharCount, pszResult, nCharCount + 1);
     165             : 
     166       49447 :         return pszResult;
     167             :     }
     168             : 
     169             :     // A few hard coded CPxxx/ISO-8859-x to UTF-8 tables
     170       27327 :     if (EQUAL(pszDstEncoding, CPL_ENC_UTF8))
     171             :     {
     172       27327 :         const auto pConvTable = CPLGetConversionTableToUTF8(pszSrcEncoding);
     173       27327 :         if (pConvTable)
     174             :         {
     175       27327 :             const auto convTable = *pConvTable;
     176       27327 :             const size_t nCharCount = strlen(pszSource);
     177             :             char *pszResult =
     178       27327 :                 static_cast<char *>(CPLCalloc(1, nCharCount * 3 + 1));
     179       27327 :             size_t iDst = 0;
     180       27327 :             unsigned char *pabyResult =
     181             :                 reinterpret_cast<unsigned char *>(pszResult);
     182      636132 :             for (size_t i = 0; i < nCharCount; ++i)
     183             :             {
     184      608805 :                 const unsigned char nChar =
     185      608805 :                     static_cast<unsigned char>(pszSource[i]);
     186      608805 :                 if (nChar <= 127)
     187             :                 {
     188      554294 :                     pszResult[iDst] = pszSource[i];
     189      554294 :                     ++iDst;
     190             :                 }
     191             :                 else
     192             :                 {
     193       54511 :                     const unsigned char nShiftedChar = nChar - 128;
     194       54511 :                     if (convTable[nShiftedChar][0])
     195             :                     {
     196       54510 :                         pabyResult[iDst] = convTable[nShiftedChar][0];
     197       54510 :                         ++iDst;
     198       54510 :                         CPLAssert(convTable[nShiftedChar][1]);
     199       54510 :                         pabyResult[iDst] = convTable[nShiftedChar][1];
     200       54510 :                         ++iDst;
     201       54510 :                         if (convTable[nShiftedChar][2])
     202             :                         {
     203          13 :                             pabyResult[iDst] = convTable[nShiftedChar][2];
     204          13 :                             ++iDst;
     205             :                         }
     206             :                     }
     207             :                     else
     208             :                     {
     209             :                         // Skip the invalid sequence in the input string.
     210           1 :                         if (!bHaveWarned2)
     211             :                         {
     212           1 :                             bHaveWarned2 = true;
     213           1 :                             CPLError(CE_Warning, CPLE_AppDefined,
     214             :                                      "One or several characters couldn't be "
     215             :                                      "converted correctly from %s to %s. "
     216             :                                      "This warning will not be emitted anymore",
     217             :                                      pszSrcEncoding, pszDstEncoding);
     218             :                         }
     219             :                     }
     220             :                 }
     221             :             }
     222             : 
     223       27327 :             pszResult[iDst] = 0;
     224       27327 :             return pszResult;
     225             :         }
     226             :     }
     227             : 
     228             : #ifdef _WIN32
     229             :     const auto MapEncodingToWindowsCodePage = [](const char *pszEncoding)
     230             :     {
     231             :         // Cf https://learn.microsoft.com/fr-fr/windows/win32/intl/code-page-identifiers
     232             :         if (STARTS_WITH(pszEncoding, "CP"))
     233             :         {
     234             :             const int nCode = atoi(pszEncoding + strlen("CP"));
     235             :             if (nCode > 0)
     236             :                 return nCode;
     237             :             else if (EQUAL(pszEncoding, "CP_OEMCP"))
     238             :                 return CP_OEMCP;
     239             :             else if (EQUAL(pszEncoding, "CP_ACP"))
     240             :                 return CP_ACP;
     241             :         }
     242             :         else if (STARTS_WITH(pszEncoding, "WINDOWS-"))
     243             :         {
     244             :             const int nCode = atoi(pszEncoding + strlen("WINDOWS-"));
     245             :             if (nCode > 0)
     246             :                 return nCode;
     247             :         }
     248             :         else if (STARTS_WITH(pszEncoding, "ISO-8859-"))
     249             :         {
     250             :             const int nCode = atoi(pszEncoding + strlen("ISO-8859-"));
     251             :             if ((nCode >= 1 && nCode <= 9) || nCode == 13 || nCode == 15)
     252             :                 return 28590 + nCode;
     253             :         }
     254             : 
     255             :         // Return a negative value, since CP_ACP = 0
     256             :         return -1;
     257             :     };
     258             : 
     259             :     /* ---------------------------------------------------------------------*/
     260             :     /*     XXX to UTF8                                                      */
     261             :     /* ---------------------------------------------------------------------*/
     262             :     if (strcmp(pszDstEncoding, CPL_ENC_UTF8) == 0)
     263             :     {
     264             :         const int nCode = MapEncodingToWindowsCodePage(pszSrcEncoding);
     265             :         if (nCode >= 0)
     266             :         {
     267             :             return CPLWin32Recode(pszSource, nCode, CP_UTF8);
     268             :         }
     269             :     }
     270             : 
     271             :     /* ---------------------------------------------------------------------*/
     272             :     /*      UTF8 to XXX                                                     */
     273             :     /* ---------------------------------------------------------------------*/
     274             :     if (strcmp(pszSrcEncoding, CPL_ENC_UTF8) == 0)
     275             :     {
     276             :         const int nCode = MapEncodingToWindowsCodePage(pszDstEncoding);
     277             :         if (nCode >= 0)
     278             :         {
     279             :             return CPLWin32Recode(pszSource, CP_UTF8, nCode);
     280             :         }
     281             :     }
     282             : #endif
     283             : 
     284             :     /* -------------------------------------------------------------------- */
     285             :     /*      Anything else to UTF-8 is treated as ISO8859-1 to UTF-8 with    */
     286             :     /*      a one-time warning.                                             */
     287             :     /* -------------------------------------------------------------------- */
     288           0 :     if (strcmp(pszDstEncoding, CPL_ENC_UTF8) == 0)
     289             :     {
     290           0 :         const int nCharCount = static_cast<int>(strlen(pszSource));
     291           0 :         char *pszResult = static_cast<char *>(CPLCalloc(1, nCharCount * 2 + 1));
     292             : 
     293           0 :         if (!bHaveWarned1)
     294             :         {
     295           0 :             bHaveWarned1 = true;
     296           0 :             CPLError(CE_Warning, CPLE_AppDefined,
     297             :                      "Recode from %s to UTF-8 not supported, "
     298             :                      "treated as ISO-8859-1 to UTF-8.",
     299             :                      pszSrcEncoding);
     300             :         }
     301             : 
     302           0 :         utf8froma(pszResult, nCharCount * 2 + 1, pszSource, nCharCount);
     303             : 
     304           0 :         return pszResult;
     305             :     }
     306             : 
     307             :     /* -------------------------------------------------------------------- */
     308             :     /*      Everything else is treated as a no-op with a warning.           */
     309             :     /* -------------------------------------------------------------------- */
     310             :     {
     311           0 :         if (!bHaveWarned3)
     312             :         {
     313           0 :             bHaveWarned3 = true;
     314           0 :             CPLError(CE_Warning, CPLE_AppDefined,
     315             :                      "Recode from %s to %s not supported, no change applied.",
     316             :                      pszSrcEncoding, pszDstEncoding);
     317             :         }
     318             : 
     319           0 :         return CPLStrdup(pszSource);
     320             :     }
     321             : }
     322             : 
     323             : /************************************************************************/
     324             : /*                       CPLRecodeFromWCharStub()                       */
     325             : /************************************************************************/
     326             : 
     327             : /**
     328             :  * Convert wchar_t string to UTF-8.
     329             :  *
     330             :  * Convert a wchar_t string into a multibyte utf-8 string.  The only
     331             :  * guaranteed supported source encoding is CPL_ENC_UCS2, and the only
     332             :  * guaranteed supported destination encodings are CPL_ENC_UTF8, CPL_ENC_ASCII
     333             :  * and CPL_ENC_ISO8859_1.  In some cases (i.e. using iconv()) other encodings
     334             :  * may also be supported.
     335             :  *
     336             :  * Note that the wchar_t type varies in size on different systems. On
     337             :  * win32 it is normally 2 bytes, and on unix 4 bytes.
     338             :  *
     339             :  * If an error occurs an error may, or may not be posted with CPLError().
     340             :  *
     341             :  * @param pwszSource the source wchar_t string, terminated with a 0 wchar_t.
     342             :  * @param pszSrcEncoding the source encoding, typically CPL_ENC_UCS2.
     343             :  * @param pszDstEncoding the destination encoding, typically CPL_ENC_UTF8.
     344             :  *
     345             :  * @return a zero terminated multi-byte string which should be freed with
     346             :  * CPLFree(), or NULL if an error occurs.
     347             :  */
     348             : 
     349      131010 : char *CPLRecodeFromWCharStub(const wchar_t *pwszSource,
     350             :                              const char *pszSrcEncoding,
     351             :                              const char *pszDstEncoding)
     352             : 
     353             : {
     354             :     /* -------------------------------------------------------------------- */
     355             :     /*      We try to avoid changes of character set.  We are just          */
     356             :     /*      providing for unicode to unicode.                               */
     357             :     /* -------------------------------------------------------------------- */
     358      131010 :     if (strcmp(pszSrcEncoding, "WCHAR_T") != 0 &&
     359      129260 :         strcmp(pszSrcEncoding, CPL_ENC_UTF8) != 0 &&
     360      129260 :         strcmp(pszSrcEncoding, CPL_ENC_UTF16) != 0 &&
     361      129260 :         strcmp(pszSrcEncoding, CPL_ENC_UCS2) != 0 &&
     362           0 :         strcmp(pszSrcEncoding, CPL_ENC_UCS4) != 0)
     363             :     {
     364           0 :         CPLError(CE_Failure, CPLE_AppDefined,
     365             :                  "Stub recoding implementation does not support "
     366             :                  "CPLRecodeFromWCharStub(...,%s,%s)",
     367             :                  pszSrcEncoding, pszDstEncoding);
     368           0 :         return nullptr;
     369             :     }
     370             : 
     371             :     /* -------------------------------------------------------------------- */
     372             :     /*      What is the source length.                                      */
     373             :     /* -------------------------------------------------------------------- */
     374      131010 :     int nSrcLen = 0;
     375             : 
     376     1933600 :     while (pwszSource[nSrcLen] != 0)
     377     1802590 :         nSrcLen++;
     378             : 
     379             :     /* -------------------------------------------------------------------- */
     380             :     /*      Allocate destination buffer plenty big.                         */
     381             :     /* -------------------------------------------------------------------- */
     382      131010 :     const int nDstBufSize = nSrcLen * 4 + 1;
     383             :     // Nearly worst case.
     384      131010 :     char *pszResult = static_cast<char *>(CPLMalloc(nDstBufSize));
     385             : 
     386      131010 :     if (nSrcLen == 0)
     387             :     {
     388       57959 :         pszResult[0] = '\0';
     389       57959 :         return pszResult;
     390             :     }
     391             : 
     392             :     /* -------------------------------------------------------------------- */
     393             :     /*      Convert, and confirm we had enough space.                       */
     394             :     /* -------------------------------------------------------------------- */
     395       73051 :     const int nDstLen = utf8fromwc(pszResult, nDstBufSize, pwszSource, nSrcLen);
     396       73051 :     if (nDstLen >= nDstBufSize)
     397             :     {
     398           0 :         CPLAssert(false);  // too small!
     399             :         return nullptr;
     400             :     }
     401             : 
     402             :     /* -------------------------------------------------------------------- */
     403             :     /*      If something other than UTF-8 was requested, recode now.        */
     404             :     /* -------------------------------------------------------------------- */
     405       73051 :     if (strcmp(pszDstEncoding, CPL_ENC_UTF8) == 0)
     406       73051 :         return pszResult;
     407             : 
     408             :     char *pszFinalResult =
     409           0 :         CPLRecodeStub(pszResult, CPL_ENC_UTF8, pszDstEncoding);
     410             : 
     411           0 :     CPLFree(pszResult);
     412             : 
     413           0 :     return pszFinalResult;
     414             : }
     415             : 
     416             : /************************************************************************/
     417             : /*                        CPLRecodeToWCharStub()                        */
     418             : /************************************************************************/
     419             : 
     420             : /**
     421             :  * Convert UTF-8 string to a wchar_t string.
     422             :  *
     423             :  * Convert a 8bit, multi-byte per character input string into a wide
     424             :  * character (wchar_t) string.  The only guaranteed supported source encodings
     425             :  * are CPL_ENC_UTF8, CPL_ENC_ASCII and CPL_ENC_ISO8869_1 (LATIN1).  The only
     426             :  * guaranteed supported destination encoding is CPL_ENC_UCS2.  Other source
     427             :  * and destination encodings may be supported depending on the underlying
     428             :  * implementation.
     429             :  *
     430             :  * Note that the wchar_t type varies in size on different systems. On
     431             :  * win32 it is normally 2 bytes, and on unix 4 bytes.
     432             :  *
     433             :  * If an error occurs an error may, or may not be posted with CPLError().
     434             :  *
     435             :  * @param pszSource input multi-byte character string.
     436             :  * @param pszSrcEncoding source encoding, typically CPL_ENC_UTF8.
     437             :  * @param pszDstEncoding destination encoding, typically CPL_ENC_UCS2.
     438             :  *
     439             :  * @return the zero terminated wchar_t string (to be freed with CPLFree()) or
     440             :  * NULL on error.
     441             :  *
     442             :  */
     443             : 
     444       41083 : wchar_t *CPLRecodeToWCharStub(const char *pszSource, const char *pszSrcEncoding,
     445             :                               const char *pszDstEncoding)
     446             : 
     447             : {
     448       41083 :     char *pszUTF8Source = const_cast<char *>(pszSource);
     449             : 
     450       41083 :     if (strcmp(pszSrcEncoding, CPL_ENC_UTF8) != 0 &&
     451           0 :         strcmp(pszSrcEncoding, CPL_ENC_ASCII) != 0)
     452             :     {
     453           0 :         pszUTF8Source = CPLRecodeStub(pszSource, pszSrcEncoding, CPL_ENC_UTF8);
     454           0 :         if (pszUTF8Source == nullptr)
     455           0 :             return nullptr;
     456             :     }
     457             : 
     458             :     /* -------------------------------------------------------------------- */
     459             :     /*      We try to avoid changes of character set.  We are just          */
     460             :     /*      providing for unicode to unicode.                               */
     461             :     /* -------------------------------------------------------------------- */
     462       41083 :     if (strcmp(pszDstEncoding, "WCHAR_T") != 0 &&
     463       41083 :         strcmp(pszDstEncoding, CPL_ENC_UCS2) != 0 &&
     464           0 :         strcmp(pszDstEncoding, CPL_ENC_UCS4) != 0 &&
     465           0 :         strcmp(pszDstEncoding, CPL_ENC_UTF16) != 0)
     466             :     {
     467           0 :         CPLError(CE_Failure, CPLE_AppDefined,
     468             :                  "Stub recoding implementation does not support "
     469             :                  "CPLRecodeToWCharStub(...,%s,%s)",
     470             :                  pszSrcEncoding, pszDstEncoding);
     471           0 :         if (pszUTF8Source != pszSource)
     472           0 :             CPLFree(pszUTF8Source);
     473           0 :         return nullptr;
     474             :     }
     475             : 
     476             :     /* -------------------------------------------------------------------- */
     477             :     /*      Do the UTF-8 to UCS-2 recoding.                                 */
     478             :     /* -------------------------------------------------------------------- */
     479       41083 :     int nSrcLen = static_cast<int>(strlen(pszUTF8Source));
     480             :     wchar_t *pwszResult =
     481       41083 :         static_cast<wchar_t *>(CPLCalloc(sizeof(wchar_t), nSrcLen + 1));
     482             : 
     483       41083 :     utf8towc(pszUTF8Source, nSrcLen, pwszResult, nSrcLen + 1);
     484             : 
     485       41083 :     if (pszUTF8Source != pszSource)
     486           0 :         CPLFree(pszUTF8Source);
     487             : 
     488       41083 :     return pwszResult;
     489             : }
     490             : 
     491             : /************************************************************************/
     492             : /*                             CPLIsUTF8()                              */
     493             : /************************************************************************/
     494             : 
     495             : /**
     496             :  * Test if a string is encoded as UTF-8.
     497             :  *
     498             :  * @param pabyData input string to test
     499             :  * @param nLen length of the input string, or -1 if the function must compute
     500             :  *             the string length. In which case it must be null terminated.
     501             :  * @return TRUE if the string is encoded as UTF-8. FALSE otherwise
     502             :  *
     503             :  */
     504       20825 : int CPLIsUTF8(const char *pabyData, int nLen)
     505             : {
     506       20825 :     if (nLen < 0)
     507       14899 :         nLen = static_cast<int>(strlen(pabyData));
     508       20825 :     return utf8test(pabyData, static_cast<unsigned>(nLen)) != 0;
     509             : }
     510             : 
     511             : /************************************************************************/
     512             : /* ==================================================================== */
     513             : /*      UTF.C code from FLTK with some modifications.                   */
     514             : /* ==================================================================== */
     515             : /************************************************************************/
     516             : 
     517             : /* Set to 1 to turn bad UTF8 bytes into ISO-8859-1. If this is to zero
     518             :    they are instead turned into the Unicode REPLACEMENT CHARACTER, of
     519             :    value 0xfffd.
     520             :    If this is on utf8decode will correctly map most (perhaps all)
     521             :    human-readable text that is in ISO-8859-1. This may allow you
     522             :    to completely ignore character sets in your code because virtually
     523             :    everything is either ISO-8859-1 or UTF-8.
     524             : */
     525             : #define ERRORS_TO_ISO8859_1 1
     526             : 
     527             : /* Set to 1 to turn bad UTF8 bytes in the 0x80-0x9f range into the
     528             :    Unicode index for Microsoft's CP1252 character set. You should
     529             :    also set ERRORS_TO_ISO8859_1. With this a huge amount of more
     530             :    available text (such as all web pages) are correctly converted
     531             :    to Unicode.
     532             : */
     533             : #define ERRORS_TO_CP1252 1
     534             : 
     535             : /* A number of Unicode code points are in fact illegal and should not
     536             :    be produced by a UTF-8 converter. Turn this on will replace the
     537             :    bytes in those encodings with errors. If you do this then converting
     538             :    arbitrary 16-bit data to UTF-8 and then back is not an identity,
     539             :    which will probably break a lot of software.
     540             : */
     541             : #define STRICT_RFC3629 0
     542             : 
     543             : #if ERRORS_TO_CP1252
     544             : // Codes 0x80..0x9f from the Microsoft CP1252 character set, translated
     545             : // to Unicode:
     546             : constexpr unsigned short cp1252[32] = {
     547             :     0x20ac, 0x0081, 0x201a, 0x0192, 0x201e, 0x2026, 0x2020, 0x2021,
     548             :     0x02c6, 0x2030, 0x0160, 0x2039, 0x0152, 0x008d, 0x017d, 0x008f,
     549             :     0x0090, 0x2018, 0x2019, 0x201c, 0x201d, 0x2022, 0x2013, 0x2014,
     550             :     0x02dc, 0x2122, 0x0161, 0x203a, 0x0153, 0x009d, 0x017e, 0x0178};
     551             : #endif
     552             : 
     553             : /************************************************************************/
     554             : /*                             utf8decode()                             */
     555             : /************************************************************************/
     556             : 
     557             : /*
     558             :     Decode a single UTF-8 encoded character starting at \e p. The
     559             :     resulting Unicode value (in the range 0-0x10ffff) is returned,
     560             :     and \e len is set the number of bytes in the UTF-8 encoding
     561             :     (adding \e len to \e p will point at the next character).
     562             : 
     563             :     If \a p points at an illegal UTF-8 encoding, including one that
     564             :     would go past \e end, or where a code is uses more bytes than
     565             :     necessary, then *reinterpret_cast<const unsigned char*>(p) is translated as
     566             : though it is in the Microsoft CP1252 character set and \e len is set to 1.
     567             :     Treating errors this way allows this to decode almost any
     568             :     ISO-8859-1 or CP1252 text that has been mistakenly placed where
     569             :     UTF-8 is expected, and has proven very useful.
     570             : 
     571             :     If you want errors to be converted to error characters (as the
     572             :     standards recommend), adding a test to see if the length is
     573             :     unexpectedly 1 will work:
     574             : 
     575             : \code
     576             :     if( *p & 0x80 )
     577             :     {  // What should be a multibyte encoding.
     578             :       code = utf8decode(p, end, &len);
     579             :       if( len<2 ) code = 0xFFFD;  // Turn errors into REPLACEMENT CHARACTER.
     580             :     }
     581             :     else
     582             :     {  // Handle the 1-byte utf8 encoding:
     583             :       code = *p;
     584             :       len = 1;
     585             :     }
     586             : \endcode
     587             : 
     588             :     Direct testing for the 1-byte case (as shown above) will also
     589             :     speed up the scanning of strings where the majority of characters
     590             :     are ASCII.
     591             : */
     592        2617 : static unsigned utf8decode(const char *p, const char *end, int *len)
     593             : {
     594        2617 :     unsigned char c = *reinterpret_cast<const unsigned char *>(p);
     595        2617 :     if (c < 0x80)
     596             :     {
     597           0 :         *len = 1;
     598           0 :         return c;
     599             : #if ERRORS_TO_CP1252
     600             :     }
     601        2617 :     else if (c < 0xa0)
     602             :     {
     603          40 :         *len = 1;
     604          40 :         return cp1252[c - 0x80];
     605             : #endif
     606             :     }
     607        2577 :     else if (c < 0xc2)
     608             :     {
     609          10 :         goto FAIL;
     610             :     }
     611        2567 :     if (p + 1 >= end || (p[1] & 0xc0) != 0x80)
     612          71 :         goto FAIL;
     613        2496 :     if (c < 0xe0)
     614             :     {
     615        2488 :         *len = 2;
     616        2488 :         return ((p[0] & 0x1f) << 6) + ((p[1] & 0x3f));
     617             :     }
     618           8 :     else if (c == 0xe0)
     619             :     {
     620           0 :         if ((reinterpret_cast<const unsigned char *>(p))[1] < 0xa0)
     621           0 :             goto FAIL;
     622           0 :         goto UTF8_3;
     623             : #if STRICT_RFC3629
     624             :     }
     625             :     else if (c == 0xed)
     626             :     {
     627             :         // RFC 3629 says surrogate chars are illegal.
     628             :         if ((reinterpret_cast<const unsigned char *>(p))[1] >= 0xa0)
     629             :             goto FAIL;
     630             :         goto UTF8_3;
     631             :     }
     632             :     else if (c == 0xef)
     633             :     {
     634             :         // 0xfffe and 0xffff are also illegal characters.
     635             :         if ((reinterpret_cast<const unsigned char *>(p))[1] == 0xbf &&
     636             :             (reinterpret_cast<const unsigned char *>(p))[2] >= 0xbe)
     637             :             goto FAIL;
     638             :         goto UTF8_3;
     639             : #endif
     640             :     }
     641           8 :     else if (c < 0xf0)
     642             :     {
     643           4 :     UTF8_3:
     644           4 :         if (p + 2 >= end || (p[2] & 0xc0) != 0x80)
     645           0 :             goto FAIL;
     646           4 :         *len = 3;
     647           4 :         return ((p[0] & 0x0f) << 12) + ((p[1] & 0x3f) << 6) + ((p[2] & 0x3f));
     648             :     }
     649           4 :     else if (c == 0xf0)
     650             :     {
     651           4 :         if ((reinterpret_cast<const unsigned char *>(p))[1] < 0x90)
     652           0 :             goto FAIL;
     653           4 :         goto UTF8_4;
     654             :     }
     655           0 :     else if (c < 0xf4)
     656             :     {
     657           0 :     UTF8_4:
     658           4 :         if (p + 3 >= end || (p[2] & 0xc0) != 0x80 || (p[3] & 0xc0) != 0x80)
     659           0 :             goto FAIL;
     660           4 :         *len = 4;
     661             : #if STRICT_RFC3629
     662             :         // RFC 3629 says all codes ending in fffe or ffff are illegal:
     663             :         if ((p[1] & 0xf) == 0xf &&
     664             :             (reinterpret_cast<const unsigned char *>(p))[2] == 0xbf &&
     665             :             (reinterpret_cast<const unsigned char *>(p))[3] >= 0xbe)
     666             :             goto FAIL;
     667             : #endif
     668           4 :         return ((p[0] & 0x07) << 18) + ((p[1] & 0x3f) << 12) +
     669           4 :                ((p[2] & 0x3f) << 6) + ((p[3] & 0x3f));
     670             :     }
     671           0 :     else if (c == 0xf4)
     672             :     {
     673           0 :         if ((reinterpret_cast<const unsigned char *>(p))[1] > 0x8f)
     674           0 :             goto FAIL;  // After 0x10ffff.
     675           0 :         goto UTF8_4;
     676             :     }
     677             :     else
     678             :     {
     679           0 :     FAIL:
     680          81 :         *len = 1;
     681             : #if ERRORS_TO_ISO8859_1
     682          81 :         return c;
     683             : #else
     684             :         return 0xfffd;  // Unicode REPLACEMENT CHARACTER
     685             : #endif
     686             :     }
     687             : }
     688             : 
     689             : /************************************************************************/
     690             : /*                              utf8towc()                              */
     691             : /************************************************************************/
     692             : 
     693             : /*  Convert a UTF-8 sequence into an array of wchar_t. These
     694             :     are used by some system calls, especially on Windows.
     695             : 
     696             :     \a src points at the UTF-8, and \a srclen is the number of bytes to
     697             :     convert.
     698             : 
     699             :     \a dst points at an array to write, and \a dstlen is the number of
     700             :     locations in this array. At most \a dstlen-1 words will be
     701             :     written there, plus a 0 terminating word. Thus this function
     702             :     will never overwrite the buffer and will always return a
     703             :     zero-terminated string. If \a dstlen is zero then \a dst can be
     704             :     null and no data is written, but the length is returned.
     705             : 
     706             :     The return value is the number of words that \e would be written
     707             :     to \a dst if it were long enough, not counting the terminating
     708             :     zero. If the return value is greater or equal to \a dstlen it
     709             :     indicates truncation, you can then allocate a new array of size
     710             :     return+1 and call this again.
     711             : 
     712             :     Errors in the UTF-8 are converted as though each byte in the
     713             :     erroneous string is in the Microsoft CP1252 encoding. This allows
     714             :     ISO-8859-1 text mistakenly identified as UTF-8 to be printed
     715             :     correctly.
     716             : 
     717             :     Notice that sizeof(wchar_t) is 2 on Windows and is 4 on Linux
     718             :     and most other systems. Where wchar_t is 16 bits, Unicode
     719             :     characters in the range 0x10000 to 0x10ffff are converted to
     720             :     "surrogate pairs" which take two words each (this is called UTF-16
     721             :     encoding). If wchar_t is 32 bits this rather nasty problem is
     722             :     avoided.
     723             : */
     724       41083 : static unsigned utf8towc(const char *src, unsigned srclen, wchar_t *dst,
     725             :                          unsigned dstlen)
     726             : {
     727       41083 :     const char *p = src;
     728       41083 :     const char *e = src + srclen;
     729       41083 :     unsigned count = 0;
     730       41083 :     if (dstlen)
     731             :         while (true)
     732             :         {
     733      300352 :             if (p >= e)
     734             :             {
     735       41083 :                 dst[count] = 0;
     736       41083 :                 return count;
     737             :             }
     738      259269 :             if (!(*p & 0x80))
     739             :             {
     740             :                 // ASCII
     741      259067 :                 dst[count] = *p++;
     742             :             }
     743             :             else
     744             :             {
     745         202 :                 int len = 0;
     746         202 :                 unsigned ucs = utf8decode(p, e, &len);
     747         202 :                 p += len;
     748             : #ifdef _WIN32
     749             :                 if (ucs < 0x10000)
     750             :                 {
     751             :                     dst[count] = static_cast<wchar_t>(ucs);
     752             :                 }
     753             :                 else
     754             :                 {
     755             :                     // Make a surrogate pair:
     756             :                     if (count + 2 >= dstlen)
     757             :                     {
     758             :                         dst[count] = 0;
     759             :                         count += 2;
     760             :                         break;
     761             :                     }
     762             :                     dst[count] = static_cast<wchar_t>(
     763             :                         (((ucs - 0x10000u) >> 10) & 0x3ff) | 0xd800);
     764             :                     dst[++count] = static_cast<wchar_t>((ucs & 0x3ff) | 0xdc00);
     765             :                 }
     766             : #else
     767         202 :                 dst[count] = static_cast<wchar_t>(ucs);
     768             : #endif
     769             :             }
     770      259269 :             if (++count == dstlen)
     771             :             {
     772           0 :                 dst[count - 1] = 0;
     773           0 :                 break;
     774             :             }
     775      259269 :         }
     776             :     // We filled dst, measure the rest:
     777           0 :     while (p < e)
     778             :     {
     779           0 :         if (!(*p & 0x80))
     780             :         {
     781           0 :             p++;
     782             :         }
     783             :         else
     784             :         {
     785           0 :             int len = 0;
     786             : #ifdef _WIN32
     787             :             const unsigned ucs = utf8decode(p, e, &len);
     788             :             p += len;
     789             :             if (ucs >= 0x10000)
     790             :                 ++count;
     791             : #else
     792           0 :             utf8decode(p, e, &len);
     793           0 :             p += len;
     794             : #endif
     795             :         }
     796           0 :         ++count;
     797             :     }
     798             : 
     799           0 :     return count;
     800             : }
     801             : 
     802             : /************************************************************************/
     803             : /*                              utf8toa()                               */
     804             : /************************************************************************/
     805             : /* Convert a UTF-8 sequence into an array of 1-byte characters.
     806             : 
     807             :     If the UTF-8 decodes to a character greater than 0xff then it is
     808             :     replaced with '?'.
     809             : 
     810             :     Errors in the UTF-8 are converted as individual bytes, same as
     811             :     utf8decode() does. This allows ISO-8859-1 text mistakenly identified
     812             :     as UTF-8 to be printed correctly (and possibly CP1252 on Windows).
     813             : 
     814             :     \a src points at the UTF-8, and \a srclen is the number of bytes to
     815             :     convert.
     816             : 
     817             :     Up to \a dstlen bytes are written to \a dst, including a null
     818             :     terminator. The return value is the number of bytes that would be
     819             :     written, not counting the null terminator. If greater or equal to
     820             :     \a dstlen then if you malloc a new array of size n+1 you will have
     821             :     the space needed for the entire string. If \a dstlen is zero then
     822             :     nothing is written and this call just measures the storage space
     823             :     needed.
     824             : */
     825       49447 : static unsigned int utf8toa(const char *src, unsigned srclen, char *dst,
     826             :                             unsigned dstlen)
     827             : {
     828       49447 :     const char *p = src;
     829       49447 :     const char *e = src + srclen;
     830       49447 :     unsigned int count = 0;
     831       49447 :     if (dstlen)
     832             :         while (true)
     833             :         {
     834      179832 :             if (p >= e)
     835             :             {
     836       49447 :                 dst[count] = 0;
     837       49447 :                 return count;
     838             :             }
     839      130385 :             unsigned char c = *reinterpret_cast<const unsigned char *>(p);
     840      130385 :             if (c < 0xC2)
     841             :             {
     842             :                 // ASCII or bad code.
     843      129585 :                 dst[count] = c;
     844      129585 :                 p++;
     845             :             }
     846             :             else
     847             :             {
     848         800 :                 int len = 0;
     849         800 :                 const unsigned int ucs = utf8decode(p, e, &len);
     850         800 :                 p += len;
     851         800 :                 if (ucs < 0x100)
     852             :                 {
     853         796 :                     dst[count] = static_cast<char>(ucs);
     854             :                 }
     855             :                 else
     856             :                 {
     857           4 :                     if (!bHaveWarned4)
     858             :                     {
     859           2 :                         bHaveWarned4 = true;
     860           2 :                         CPLError(
     861             :                             CE_Warning, CPLE_AppDefined,
     862             :                             "One or several characters couldn't be converted "
     863             :                             "correctly from UTF-8 to ISO-8859-1.  "
     864             :                             "This warning will not be emitted anymore.");
     865             :                     }
     866           4 :                     dst[count] = '?';
     867             :                 }
     868             :             }
     869      130385 :             if (++count >= dstlen)
     870             :             {
     871           0 :                 dst[count - 1] = 0;
     872           0 :                 break;
     873             :             }
     874      130385 :         }
     875             :     // We filled dst, measure the rest:
     876           0 :     while (p < e)
     877             :     {
     878           0 :         if (!(*p & 0x80))
     879             :         {
     880           0 :             p++;
     881             :         }
     882             :         else
     883             :         {
     884           0 :             int len = 0;
     885           0 :             utf8decode(p, e, &len);
     886           0 :             p += len;
     887             :         }
     888           0 :         ++count;
     889             :     }
     890           0 :     return count;
     891             : }
     892             : 
     893             : /************************************************************************/
     894             : /*                             utf8fromwc()                             */
     895             : /************************************************************************/
     896             : /* Turn "wide characters" as returned by some system calls
     897             :     (especially on Windows) into UTF-8.
     898             : 
     899             :     Up to \a dstlen bytes are written to \a dst, including a null
     900             :     terminator. The return value is the number of bytes that would be
     901             :     written, not counting the null terminator. If greater or equal to
     902             :     \a dstlen then if you malloc a new array of size n+1 you will have
     903             :     the space needed for the entire string. If \a dstlen is zero then
     904             :     nothing is written and this call just measures the storage space
     905             :     needed.
     906             : 
     907             :     \a srclen is the number of words in \a src to convert. On Windows
     908             :     this is not necessarily the number of characters, due to there
     909             :     possibly being "surrogate pairs" in the UTF-16 encoding used.
     910             :     On Unix wchar_t is 32 bits and each location is a character.
     911             : 
     912             :     On Unix if a src word is greater than 0x10ffff then this is an
     913             :     illegal character according to RFC 3629. These are converted as
     914             :     though they are 0xFFFD (REPLACEMENT CHARACTER). Characters in the
     915             :     range 0xd800 to 0xdfff, or ending with 0xfffe or 0xffff are also
     916             :     illegal according to RFC 3629. However I encode these as though
     917             :     they are legal, so that utf8towc will return the original data.
     918             : 
     919             :     On Windows "surrogate pairs" are converted to a single character
     920             :     and UTF-8 encoded (as 4 bytes). Mismatched halves of surrogate
     921             :     pairs are converted as though they are individual characters.
     922             : */
     923       73051 : static unsigned int utf8fromwc(char *dst, unsigned dstlen, const wchar_t *src,
     924             :                                unsigned srclen)
     925             : {
     926       73051 :     unsigned int i = 0;
     927       73051 :     unsigned int count = 0;
     928       73051 :     if (dstlen)
     929             :         while (true)
     930             :         {
     931     1875640 :             if (i >= srclen)
     932             :             {
     933       73051 :                 dst[count] = 0;
     934       73051 :                 return count;
     935             :             }
     936     1802590 :             unsigned int ucs = src[i++];
     937     1802590 :             if (ucs < 0x80U)
     938             :             {
     939     1795580 :                 dst[count++] = static_cast<char>(ucs);
     940     1795580 :                 if (count >= dstlen)
     941             :                 {
     942           0 :                     dst[count - 1] = 0;
     943           0 :                     break;
     944             :                 }
     945             :             }
     946        7013 :             else if (ucs < 0x800U)
     947             :             {
     948             :                 // 2 bytes.
     949        4370 :                 if (count + 2 >= dstlen)
     950             :                 {
     951           0 :                     dst[count] = 0;
     952           0 :                     count += 2;
     953           0 :                     break;
     954             :                 }
     955        4370 :                 dst[count++] = 0xc0 | static_cast<char>(ucs >> 6);
     956        4370 :                 dst[count++] = 0x80 | static_cast<char>(ucs & 0x3F);
     957             : #ifdef _WIN32
     958             :             }
     959             :             else if (ucs >= 0xd800 && ucs <= 0xdbff && i < srclen &&
     960             :                      src[i] >= 0xdc00 && src[i] <= 0xdfff)
     961             :             {
     962             :                 // Surrogate pair.
     963             :                 unsigned int ucs2 = src[i++];
     964             :                 ucs = 0x10000U + ((ucs & 0x3ff) << 10) + (ucs2 & 0x3ff);
     965             :             // All surrogate pairs turn into 4-byte utf8.
     966             : #else
     967             :             }
     968        2643 :             else if (ucs >= 0x10000)
     969             :             {
     970           1 :                 if (ucs > 0x10ffff)
     971             :                 {
     972           1 :                     ucs = 0xfffd;
     973           1 :                     goto J1;
     974             :                 }
     975             : #endif
     976           0 :                 if (count + 4 >= dstlen)
     977             :                 {
     978           0 :                     dst[count] = 0;
     979           0 :                     count += 4;
     980           0 :                     break;
     981             :                 }
     982           0 :                 dst[count++] = 0xf0 | static_cast<char>(ucs >> 18);
     983           0 :                 dst[count++] = 0x80 | static_cast<char>((ucs >> 12) & 0x3F);
     984           0 :                 dst[count++] = 0x80 | static_cast<char>((ucs >> 6) & 0x3F);
     985           0 :                 dst[count++] = 0x80 | static_cast<char>(ucs & 0x3F);
     986             :             }
     987             :             else
     988             :             {
     989             : #ifndef _WIN32
     990        2642 :             J1:
     991             : #endif
     992             :                 // All others are 3 bytes:
     993        2643 :                 if (count + 3 >= dstlen)
     994             :                 {
     995           0 :                     dst[count] = 0;
     996           0 :                     count += 3;
     997           0 :                     break;
     998             :                 }
     999        2643 :                 dst[count++] = 0xe0 | static_cast<char>(ucs >> 12);
    1000        2643 :                 dst[count++] = 0x80 | static_cast<char>((ucs >> 6) & 0x3F);
    1001        2643 :                 dst[count++] = 0x80 | static_cast<char>(ucs & 0x3F);
    1002             :             }
    1003     1802590 :         }
    1004             : 
    1005             :     // We filled dst, measure the rest:
    1006           0 :     while (i < srclen)
    1007             :     {
    1008           0 :         unsigned int ucs = src[i++];
    1009           0 :         if (ucs < 0x80U)
    1010             :         {
    1011           0 :             count++;
    1012             :         }
    1013           0 :         else if (ucs < 0x800U)
    1014             :         {
    1015             :             // 2 bytes.
    1016           0 :             count += 2;
    1017             : #ifdef _WIN32
    1018             :         }
    1019             :         else if (ucs >= 0xd800 && ucs <= 0xdbff && i < srclen - 1 &&
    1020             :                  src[i + 1] >= 0xdc00 && src[i + 1] <= 0xdfff)
    1021             :         {
    1022             :             // Surrogate pair.
    1023             :             ++i;
    1024             : #else
    1025             :         }
    1026           0 :         else if (ucs >= 0x10000 && ucs <= 0x10ffff)
    1027             :         {
    1028             : #endif
    1029           0 :             count += 4;
    1030             :         }
    1031             :         else
    1032             :         {
    1033           0 :             count += 3;
    1034             :         }
    1035             :     }
    1036           0 :     return count;
    1037             : }
    1038             : 
    1039             : /************************************************************************/
    1040             : /*                             utf8froma()                              */
    1041             : /************************************************************************/
    1042             : 
    1043             : /* Convert an ISO-8859-1 (i.e. normal c-string) byte stream to UTF-8.
    1044             : 
    1045             :     It is possible this should convert Microsoft's CP1252 to UTF-8
    1046             :     instead. This would translate the codes in the range 0x80-0x9f
    1047             :     to different characters. Currently it does not do this.
    1048             : 
    1049             :     Up to \a dstlen bytes are written to \a dst, including a null
    1050             :     terminator. The return value is the number of bytes that would be
    1051             :     written, not counting the null terminator. If greater or equal to
    1052             :     \a dstlen then if you malloc a new array of size n+1 you will have
    1053             :     the space needed for the entire string. If \a dstlen is zero then
    1054             :     nothing is written and this call just measures the storage space
    1055             :     needed.
    1056             : 
    1057             :     \a srclen is the number of bytes in \a src to convert.
    1058             : 
    1059             :     If the return value equals \a srclen then this indicates that
    1060             :     no conversion is necessary, as only ASCII characters are in the
    1061             :     string.
    1062             : */
    1063     1958380 : static unsigned utf8froma(char *dst, unsigned dstlen, const char *src,
    1064             :                           unsigned srclen)
    1065             : {
    1066     1958380 :     const char *p = src;
    1067     1958380 :     const char *e = src + srclen;
    1068     1958380 :     unsigned count = 0;
    1069     1958380 :     if (dstlen)
    1070             :         while (true)
    1071             :         {
    1072    31407200 :             if (p >= e)
    1073             :             {
    1074     1958380 :                 dst[count] = 0;
    1075     1958380 :                 return count;
    1076             :             }
    1077    29448800 :             unsigned char ucs = *reinterpret_cast<const unsigned char *>(p);
    1078    29448800 :             p++;
    1079    29448800 :             if (ucs < 0x80U)
    1080             :             {
    1081    29394500 :                 dst[count++] = ucs;
    1082    29394500 :                 if (count >= dstlen)
    1083             :                 {
    1084           0 :                     dst[count - 1] = 0;
    1085           0 :                     break;
    1086             :                 }
    1087             :             }
    1088             :             else
    1089             :             {
    1090             :                 // 2 bytes (note that CP1252 translate could make 3 bytes!)
    1091       54280 :                 if (count + 2 >= dstlen)
    1092             :                 {
    1093           0 :                     dst[count] = 0;
    1094           0 :                     count += 2;
    1095           0 :                     break;
    1096             :                 }
    1097       54280 :                 dst[count++] = 0xc0 | (ucs >> 6);
    1098       54280 :                 dst[count++] = 0x80 | (ucs & 0x3F);
    1099             :             }
    1100    29448800 :         }
    1101             : 
    1102             :     // We filled dst, measure the rest:
    1103           0 :     while (p < e)
    1104             :     {
    1105           0 :         unsigned char ucs = *reinterpret_cast<const unsigned char *>(p);
    1106           0 :         p++;
    1107           0 :         if (ucs < 0x80U)
    1108             :         {
    1109           0 :             count++;
    1110             :         }
    1111             :         else
    1112             :         {
    1113           0 :             count += 2;
    1114             :         }
    1115             :     }
    1116             : 
    1117           0 :     return count;
    1118             : }
    1119             : 
    1120             : #ifdef _WIN32
    1121             : 
    1122             : /************************************************************************/
    1123             : /*                           CPLWin32Recode()                           */
    1124             : /************************************************************************/
    1125             : 
    1126             : /* Convert an CODEPAGE (i.e. normal c-string) byte stream
    1127             :      to another CODEPAGE (i.e. normal c-string) byte stream.
    1128             : 
    1129             :     \a src is target c-string byte stream (including a null terminator).
    1130             :     \a src_code_page is target c-string byte code page.
    1131             :     \a dst_code_page is destination c-string byte code page.
    1132             : 
    1133             :    UTF7          65000
    1134             :    UTF8          65001
    1135             :    OEM-US          437
    1136             :    OEM-ALABIC      720
    1137             :    OEM-GREEK       737
    1138             :    OEM-BALTIC      775
    1139             :    OEM-MLATIN1     850
    1140             :    OEM-LATIN2      852
    1141             :    OEM-CYRILLIC    855
    1142             :    OEM-TURKISH     857
    1143             :    OEM-MLATIN1P    858
    1144             :    OEM-HEBREW      862
    1145             :    OEM-RUSSIAN     866
    1146             : 
    1147             :    THAI            874
    1148             :    SJIS            932
    1149             :    GBK             936
    1150             :    KOREA           949
    1151             :    BIG5            950
    1152             : 
    1153             :    EUROPE         1250
    1154             :    CYRILLIC       1251
    1155             :    LATIN1         1252
    1156             :    GREEK          1253
    1157             :    TURKISH        1254
    1158             :    HEBREW         1255
    1159             :    ARABIC         1256
    1160             :    BALTIC         1257
    1161             :    VIETNAM        1258
    1162             : 
    1163             :    ISO-LATIN1    28591
    1164             :    ISO-LATIN2    28592
    1165             :    ISO-LATIN3    28593
    1166             :    ISO-BALTIC    28594
    1167             :    ISO-CYRILLIC  28595
    1168             :    ISO-ARABIC    28596
    1169             :    ISO-HEBREW    28598
    1170             :    ISO-TURKISH   28599
    1171             :    ISO-LATIN9    28605
    1172             : 
    1173             :    ISO-2022-JP   50220
    1174             : 
    1175             : */
    1176             : 
    1177             : char *CPLWin32Recode(const char *src, unsigned src_code_page,
    1178             :                      unsigned dst_code_page)
    1179             : {
    1180             :     // Convert from source code page to Unicode.
    1181             : 
    1182             :     // Compute the length in wide characters.
    1183             :     int wlen = MultiByteToWideChar(src_code_page, MB_ERR_INVALID_CHARS, src, -1,
    1184             :                                    nullptr, 0);
    1185             :     if (wlen == 0 && GetLastError() == ERROR_NO_UNICODE_TRANSLATION)
    1186             :     {
    1187             :         if (!bHaveWarned5)
    1188             :         {
    1189             :             bHaveWarned5 = true;
    1190             :             CPLError(
    1191             :                 CE_Warning, CPLE_AppDefined,
    1192             :                 "One or several characters could not be translated from CP%d. "
    1193             :                 "This warning will not be emitted anymore.",
    1194             :                 src_code_page);
    1195             :         }
    1196             : 
    1197             :         // Retry now without MB_ERR_INVALID_CHARS flag.
    1198             :         wlen = MultiByteToWideChar(src_code_page, 0, src, -1, nullptr, 0);
    1199             :     }
    1200             : 
    1201             :     // Do the actual conversion.
    1202             :     wchar_t *tbuf =
    1203             :         static_cast<wchar_t *>(CPLCalloc(sizeof(wchar_t), wlen + 1));
    1204             :     tbuf[wlen] = 0;
    1205             :     MultiByteToWideChar(src_code_page, 0, src, -1, tbuf, wlen + 1);
    1206             : 
    1207             :     // Convert from Unicode to destination code page.
    1208             : 
    1209             :     // Compute the length in chars.
    1210             :     BOOL bUsedDefaultChar = FALSE;
    1211             :     int len = 0;
    1212             :     if (dst_code_page == CP_UTF7 || dst_code_page == CP_UTF8)
    1213             :         len = WideCharToMultiByte(dst_code_page, 0, tbuf, -1, nullptr, 0,
    1214             :                                   nullptr, nullptr);
    1215             :     else
    1216             :         len = WideCharToMultiByte(dst_code_page, 0, tbuf, -1, nullptr, 0,
    1217             :                                   nullptr, &bUsedDefaultChar);
    1218             :     if (bUsedDefaultChar)
    1219             :     {
    1220             :         if (!bHaveWarned6)
    1221             :         {
    1222             :             bHaveWarned6 = true;
    1223             :             CPLError(
    1224             :                 CE_Warning, CPLE_AppDefined,
    1225             :                 "One or several characters could not be translated to CP%d. "
    1226             :                 "This warning will not be emitted anymore.",
    1227             :                 dst_code_page);
    1228             :         }
    1229             :     }
    1230             : 
    1231             :     // Do the actual conversion.
    1232             :     char *pszResult = static_cast<char *>(CPLCalloc(sizeof(char), len + 1));
    1233             :     WideCharToMultiByte(dst_code_page, 0, tbuf, -1, pszResult, len + 1, nullptr,
    1234             :                         nullptr);
    1235             :     pszResult[len] = 0;
    1236             : 
    1237             :     CPLFree(tbuf);
    1238             : 
    1239             :     return pszResult;
    1240             : }
    1241             : 
    1242             : #endif
    1243             : 
    1244             : /*
    1245             : ** For now we disable the rest which is locale() related.  We may need
    1246             : ** parts of it later.
    1247             : */
    1248             : 
    1249             : #ifdef notdef
    1250             : 
    1251             : #ifdef _WIN32
    1252             : #include <windows.h>
    1253             : #endif
    1254             : 
    1255             : /*! Return true if the "locale" seems to indicate that UTF-8 encoding
    1256             :     is used. If true the utf8tomb and utf8frommb don't do anything
    1257             :     useful.
    1258             : 
    1259             :     <i>It is highly recommended that you change your system so this
    1260             :     does return true.</i> On Windows this is done by setting the
    1261             :     "codepage" to CP_UTF8.  On Unix this is done by setting $LC_CTYPE
    1262             :     to a string containing the letters "utf" or "UTF" in it, or by
    1263             :     deleting all $LC* and $LANG environment variables. In the future
    1264             :     it is likely that all non-Asian Unix systems will return true,
    1265             :     due to the compatibility of UTF-8 with ISO-8859-1.
    1266             : */
    1267             : int utf8locale(void)
    1268             : {
    1269             :     static int ret = 2;
    1270             :     if (ret == 2)
    1271             :     {
    1272             : #ifdef _WIN32
    1273             :         ret = GetACP() == CP_UTF8;
    1274             : #else
    1275             :         char *s;
    1276             :         ret = 1;  // assume UTF-8 if no locale
    1277             :         if (((s = getenv("LC_CTYPE")) && *s) ||
    1278             :             ((s = getenv("LC_ALL")) && *s) || ((s = getenv("LANG")) && *s))
    1279             :         {
    1280             :             ret = strstr(s, "utf") || strstr(s, "UTF");
    1281             :         }
    1282             : #endif
    1283             :     }
    1284             : 
    1285             :     return ret;
    1286             : }
    1287             : 
    1288             : /*! Convert the UTF-8 used by FLTK to the locale-specific encoding
    1289             :     used for filenames (and sometimes used for data in files).
    1290             :     Unfortunately due to stupid design you will have to do this as
    1291             :     needed for filenames. This is a bug on both Unix and Windows.
    1292             : 
    1293             :     Up to \a dstlen bytes are written to \a dst, including a null
    1294             :     terminator. The return value is the number of bytes that would be
    1295             :     written, not counting the null terminator. If greater or equal to
    1296             :     \a dstlen then if you malloc a new array of size n+1 you will have
    1297             :     the space needed for the entire string. If \a dstlen is zero then
    1298             :     nothing is written and this call just measures the storage space
    1299             :     needed.
    1300             : 
    1301             :     If utf8locale() returns true then this does not change the data.
    1302             :     It is copied and truncated as necessary to
    1303             :     the destination buffer and \a srclen is always returned.  */
    1304             : unsigned utf8tomb(const char *src, unsigned srclen, char *dst, unsigned dstlen)
    1305             : {
    1306             :     if (!utf8locale())
    1307             :     {
    1308             : #ifdef _WIN32
    1309             :         wchar_t lbuf[1024] = {};
    1310             :         wchar_t *buf = lbuf;
    1311             :         unsigned length = utf8towc(src, srclen, buf, 1024);
    1312             :         unsigned ret;
    1313             :         if (length >= 1024)
    1314             :         {
    1315             :             buf =
    1316             :                 static_cast<wchar_t *>(malloc((length + 1) * sizeof(wchar_t)));
    1317             :             utf8towc(src, srclen, buf, length + 1);
    1318             :         }
    1319             :         if (dstlen)
    1320             :         {
    1321             :             // apparently this does not null-terminate, even though msdn
    1322             :             // documentation claims it does:
    1323             :             ret = WideCharToMultiByte(GetACP(), 0, buf, length, dst, dstlen, 0,
    1324             :                                       0);
    1325             :             dst[ret] = 0;
    1326             :         }
    1327             :         // if it overflows or measuring length, get the actual length:
    1328             :         if (dstlen == 0 || ret >= dstlen - 1)
    1329             :             ret = WideCharToMultiByte(GetACP(), 0, buf, length, 0, 0, 0, 0);
    1330             :         if (buf != lbuf)
    1331             :             free((void *)buf);
    1332             :         return ret;
    1333             : #else
    1334             :         wchar_t lbuf[1024] = {};
    1335             :         wchar_t *buf = lbuf;
    1336             :         unsigned length = utf8towc(src, srclen, buf, 1024);
    1337             :         if (length >= 1024)
    1338             :         {
    1339             :             buf =
    1340             :                 static_cast<wchar_t *>(malloc((length + 1) * sizeof(wchar_t)));
    1341             :             utf8towc(src, srclen, buf, length + 1);
    1342             :         }
    1343             :         int ret = 0;
    1344             :         if (dstlen)
    1345             :         {
    1346             :             ret = wcstombs(dst, buf, dstlen);
    1347             :             if (ret >= dstlen - 1)
    1348             :                 ret = wcstombs(0, buf, 0);
    1349             :         }
    1350             :         else
    1351             :         {
    1352             :             ret = wcstombs(0, buf, 0);
    1353             :         }
    1354             :         if (buf != lbuf)
    1355             :             free((void *)buf);
    1356             :         if (ret >= 0)
    1357             :             return (unsigned)ret;
    1358             :         // On any errors we return the UTF-8 as raw text...
    1359             : #endif
    1360             :     }
    1361             :     // Identity transform:
    1362             :     if (srclen < dstlen)
    1363             :     {
    1364             :         memcpy(dst, src, srclen);
    1365             :         dst[srclen] = 0;
    1366             :     }
    1367             :     else
    1368             :     {
    1369             :         memcpy(dst, src, dstlen - 1);
    1370             :         dst[dstlen - 1] = 0;
    1371             :     }
    1372             :     return srclen;
    1373             : }
    1374             : 
    1375             : /*! Convert a filename from the locale-specific multibyte encoding
    1376             :     used by Windows to UTF-8 as used by FLTK.
    1377             : 
    1378             :     Up to \a dstlen bytes are written to \a dst, including a null
    1379             :     terminator. The return value is the number of bytes that would be
    1380             :     written, not counting the null terminator. If greater or equal to
    1381             :     \a dstlen then if you malloc a new array of size n+1 you will have
    1382             :     the space needed for the entire string. If \a dstlen is zero then
    1383             :     nothing is written and this call just measures the storage space
    1384             :     needed.
    1385             : 
    1386             :     On Unix or on Windows when a UTF-8 locale is in effect, this
    1387             :     does not change the data. It is copied and truncated as necessary to
    1388             :     the destination buffer and \a srclen is always returned.
    1389             :     You may also want to check if utf8test() returns non-zero, so that
    1390             :     the filesystem can store filenames in UTF-8 encoding regardless of
    1391             :     the locale.
    1392             : */
    1393             : unsigned utf8frommb(char *dst, unsigned dstlen, const char *src,
    1394             :                     unsigned srclen)
    1395             : {
    1396             :     if (!utf8locale())
    1397             :     {
    1398             : #ifdef _WIN32
    1399             :         wchar_t lbuf[1024] = {};
    1400             :         wchar_t *buf = lbuf;
    1401             :         unsigned ret;
    1402             :         const unsigned length =
    1403             :             MultiByteToWideChar(GetACP(), 0, src, srclen, buf, 1024);
    1404             :         if (length >= 1024)
    1405             :         {
    1406             :             length = MultiByteToWideChar(GetACP(), 0, src, srclen, 0, 0);
    1407             :             buf = static_cast<wchar_t *>(malloc(length * sizeof(wchar_t)));
    1408             :             MultiByteToWideChar(GetACP(), 0, src, srclen, buf, length);
    1409             :         }
    1410             :         ret = utf8fromwc(dst, dstlen, buf, length);
    1411             :         if (buf != lbuf)
    1412             :             free(buf);
    1413             :         return ret;
    1414             : #else
    1415             :         wchar_t lbuf[1024] = {};
    1416             :         wchar_t *buf = lbuf;
    1417             :         const int length = mbstowcs(buf, src, 1024);
    1418             :         if (length >= 1024)
    1419             :         {
    1420             :             length = mbstowcs(0, src, 0) + 1;
    1421             :             buf =
    1422             :                 static_cast<wchar_t *>(malloc(length * sizeof(unsigned short)));
    1423             :             mbstowcs(buf, src, length);
    1424             :         }
    1425             :         if (length >= 0)
    1426             :         {
    1427             :             const unsigned ret = utf8fromwc(dst, dstlen, buf, length);
    1428             :             if (buf != lbuf)
    1429             :                 free(buf);
    1430             :             return ret;
    1431             :         }
    1432             :         // Errors in conversion return the UTF-8 unchanged.
    1433             : #endif
    1434             :     }
    1435             :     // Identity transform:
    1436             :     if (srclen < dstlen)
    1437             :     {
    1438             :         memcpy(dst, src, srclen);
    1439             :         dst[srclen] = 0;
    1440             :     }
    1441             :     else
    1442             :     {
    1443             :         memcpy(dst, src, dstlen - 1);
    1444             :         dst[dstlen - 1] = 0;
    1445             :     }
    1446             :     return srclen;
    1447             : }
    1448             : 
    1449             : #endif  // def notdef - disabled locale specific stuff.
    1450             : 
    1451             : /*! Examines the first \a srclen bytes in \a src and return a verdict
    1452             :     on whether it is UTF-8 or not.
    1453             :     - Returns 0 if there is any illegal UTF-8 sequences, using the
    1454             :       same rules as utf8decode(). Note that some UCS values considered
    1455             :       illegal by RFC 3629, such as 0xffff, are considered legal by this.
    1456             :     - Returns 1 if there are only single-byte characters (i.e. no bytes
    1457             :       have the high bit set). This is legal UTF-8, but also indicates
    1458             :       plain ASCII. It also returns 1 if \a srclen is zero.
    1459             :     - Returns 2 if there are only characters less than 0x800.
    1460             :     - Returns 3 if there are only characters less than 0x10000.
    1461             :     - Returns 4 if there are characters in the 0x10000 to 0x10ffff range.
    1462             : 
    1463             :     Because there are many illegal sequences in UTF-8, it is almost
    1464             :     impossible for a string in another encoding to be confused with
    1465             :     UTF-8. This is very useful for transitioning Unix to UTF-8
    1466             :     filenames, you can simply test each filename with this to decide
    1467             :     if it is UTF-8 or in the locale encoding. My hope is that if
    1468             :     this is done we will be able to cleanly transition to a locale-less
    1469             :     encoding.
    1470             : */
    1471             : 
    1472       20825 : static int utf8test(const char *src, unsigned srclen)
    1473             : {
    1474       20825 :     int ret = 1;
    1475       20825 :     const char *p = src;
    1476       20825 :     const char *e = src + srclen;
    1477     1785900 :     while (p < e)
    1478             :     {
    1479     1765130 :         if (*p == 0)
    1480           0 :             return 0;
    1481     1765130 :         if (*p & 0x80)
    1482             :         {
    1483        1615 :             int len = 0;
    1484        1615 :             utf8decode(p, e, &len);
    1485        1615 :             if (len < 2)
    1486          55 :                 return 0;
    1487        1560 :             if (len > ret)
    1488         555 :                 ret = len;
    1489        1560 :             p += len;
    1490             :         }
    1491             :         else
    1492             :         {
    1493     1763510 :             p++;
    1494             :         }
    1495             :     }
    1496       20770 :     return ret;
    1497             : }

Generated by: LCOV version 1.14