Line data Source code
1 : /**********************************************************************
2 : *
3 : * Name: cpl_recode_stub.cpp
4 : * Project: CPL - Common Portability Library
5 : * Purpose: Character set recoding and char/wchar_t conversions, stub
6 : * implementation to be used if iconv() functionality is not
7 : * available.
8 : * Author: Frank Warmerdam, warmerdam@pobox.com
9 : *
10 : * The bulk of this code is derived from the utf.c module from FLTK. It
11 : * was originally downloaded from:
12 : * http://svn.easysw.com/public/fltk/fltk/trunk/src/utf.c
13 : *
14 : **********************************************************************
15 : * Copyright (c) 2008, Frank Warmerdam
16 : * Copyright 2006 by Bill Spitzak and others.
17 : * Copyright (c) 2009-2014, Even Rouault <even dot rouault at spatialys.com>
18 : *
19 : * Permission to use, copy, modify, and distribute this software for any
20 : * purpose with or without fee is hereby granted, provided that the above
21 : * copyright notice and this permission notice appear in all copies.
22 : *
23 : * THE SOFTWARE IS PROVIDED "AS IS" AND THE AUTHOR DISCLAIMS ALL WARRANTIES
24 : * WITH REGARD TO THIS SOFTWARE INCLUDING ALL IMPLIED WARRANTIES OF
25 : * MERCHANTABILITY AND FITNESS. IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR
26 : * ANY SPECIAL, DIRECT, INDIRECT, OR CONSEQUENTIAL DAMAGES OR ANY DAMAGES
27 : * WHATSOEVER RESULTING FROM LOSS OF USE, DATA OR PROFITS, WHETHER IN AN
28 : * ACTION OF CONTRACT, NEGLIGENCE OR OTHER TORTIOUS ACTION, ARISING OUT OF
29 : * OR IN CONNECTION WITH THE USE OR PERFORMANCE OF THIS SOFTWARE.
30 : **********************************************************************/
31 :
32 : #include "cpl_port.h"
33 : #include "cpl_string.h"
34 :
35 : #include <cstring>
36 :
37 : #include "cpl_conv.h"
38 : #include "cpl_error.h"
39 : #include "cpl_character_sets.c"
40 :
41 : static unsigned utf8decode(const char *p, const char *end, int *len);
42 : static unsigned utf8towc(const char *src, unsigned srclen, wchar_t *dst,
43 : unsigned dstlen);
44 : static unsigned utf8toa(const char *src, unsigned srclen, char *dst,
45 : unsigned dstlen);
46 : static unsigned utf8fromwc(char *dst, unsigned dstlen, const wchar_t *src,
47 : unsigned srclen);
48 : static unsigned utf8froma(char *dst, unsigned dstlen, const char *src,
49 : unsigned srclen);
50 : static int utf8test(const char *src, unsigned srclen);
51 :
52 : #ifdef _WIN32
53 :
54 : #include <windows.h>
55 : #include <winnls.h>
56 :
57 : static char *CPLWin32Recode(const char *src, unsigned src_code_page,
58 : unsigned dst_code_page) CPL_RETURNS_NONNULL;
59 : #endif
60 :
61 : /* used by cpl_recode.cpp */
62 : extern void CPLClearRecodeStubWarningFlags();
63 : extern char *CPLRecodeStub(const char *, const char *,
64 : const char *) CPL_RETURNS_NONNULL;
65 : extern char *CPLRecodeFromWCharStub(const wchar_t *, const char *,
66 : const char *);
67 : extern wchar_t *CPLRecodeToWCharStub(const char *, const char *, const char *);
68 :
69 : /************************************************************************/
70 : /* ==================================================================== */
71 : /* Stub Implementation not depending on iconv() or WIN32 API. */
72 : /* ==================================================================== */
73 : /************************************************************************/
74 :
75 : static bool bHaveWarned1 = false;
76 : static bool bHaveWarned2 = false;
77 : static bool bHaveWarned3 = false;
78 : static bool bHaveWarned4 = false;
79 : #ifdef _WIN32
80 : static bool bHaveWarned5 = false;
81 : static bool bHaveWarned6 = false;
82 : #endif
83 :
84 : /************************************************************************/
85 : /* CPLClearRecodeStubWarningFlags() */
86 : /************************************************************************/
87 :
88 13461 : void CPLClearRecodeStubWarningFlags()
89 : {
90 13461 : bHaveWarned1 = false;
91 13461 : bHaveWarned2 = false;
92 13461 : bHaveWarned3 = false;
93 13461 : bHaveWarned4 = false;
94 : #ifdef _WIN32
95 : bHaveWarned5 = false;
96 : bHaveWarned6 = false;
97 : #endif
98 13461 : }
99 :
100 : /************************************************************************/
101 : /* CPLRecodeStub() */
102 : /************************************************************************/
103 :
104 : /**
105 : * Convert a string from a source encoding to a destination encoding.
106 : *
107 : * The only guaranteed supported encodings are CPL_ENC_UTF8, CPL_ENC_ASCII
108 : * and CPL_ENC_ISO8859_1. Currently, the following conversions are supported :
109 : * <ul>
110 : * <li>CPL_ENC_ASCII -> CPL_ENC_UTF8 or CPL_ENC_ISO8859_1 (no conversion in
111 : * fact)</li>
112 : * <li>CPL_ENC_ISO8859_1 -> CPL_ENC_UTF8</li>
113 : * <li>CPL_ENC_UTF8 -> CPL_ENC_ISO8859_1</li>
114 : * </ul>
115 : *
116 : * If an error occurs an error may, or may not be posted with CPLError().
117 : *
118 : * @param pszSource a NULL terminated string.
119 : * @param pszSrcEncoding the source encoding.
120 : * @param pszDstEncoding the destination encoding.
121 : *
122 : * @return a NULL terminated string which should be freed with CPLFree().
123 : */
124 :
125 2035160 : char *CPLRecodeStub(const char *pszSource, const char *pszSrcEncoding,
126 : const char *pszDstEncoding)
127 :
128 : {
129 : /* -------------------------------------------------------------------- */
130 : /* If the source or destination is current locale(), we change */
131 : /* it to ISO8859-1 since our stub implementation does not */
132 : /* attempt to address locales properly. */
133 : /* -------------------------------------------------------------------- */
134 :
135 2035160 : if (pszSrcEncoding[0] == '\0')
136 0 : pszSrcEncoding = CPL_ENC_ISO8859_1;
137 :
138 2035160 : if (pszDstEncoding[0] == '\0')
139 0 : pszDstEncoding = CPL_ENC_ISO8859_1;
140 :
141 : /* -------------------------------------------------------------------- */
142 : /* ISO8859 to UTF8 */
143 : /* -------------------------------------------------------------------- */
144 2035160 : if (strcmp(pszSrcEncoding, CPL_ENC_ISO8859_1) == 0 &&
145 1958380 : strcmp(pszDstEncoding, CPL_ENC_UTF8) == 0)
146 : {
147 1958380 : const int nCharCount = static_cast<int>(strlen(pszSource));
148 1958380 : char *pszResult = static_cast<char *>(CPLCalloc(1, nCharCount * 2 + 1));
149 :
150 1958380 : utf8froma(pszResult, nCharCount * 2 + 1, pszSource, nCharCount);
151 :
152 1958380 : return pszResult;
153 : }
154 :
155 : /* -------------------------------------------------------------------- */
156 : /* UTF8 to ISO8859 */
157 : /* -------------------------------------------------------------------- */
158 76774 : if (strcmp(pszSrcEncoding, CPL_ENC_UTF8) == 0 &&
159 49447 : strcmp(pszDstEncoding, CPL_ENC_ISO8859_1) == 0)
160 : {
161 49447 : int nCharCount = static_cast<int>(strlen(pszSource));
162 49447 : char *pszResult = static_cast<char *>(CPLCalloc(1, nCharCount + 1));
163 :
164 49447 : utf8toa(pszSource, nCharCount, pszResult, nCharCount + 1);
165 :
166 49447 : return pszResult;
167 : }
168 :
169 : // A few hard coded CPxxx/ISO-8859-x to UTF-8 tables
170 27327 : if (EQUAL(pszDstEncoding, CPL_ENC_UTF8))
171 : {
172 27327 : const auto pConvTable = CPLGetConversionTableToUTF8(pszSrcEncoding);
173 27327 : if (pConvTable)
174 : {
175 27327 : const auto convTable = *pConvTable;
176 27327 : const size_t nCharCount = strlen(pszSource);
177 : char *pszResult =
178 27327 : static_cast<char *>(CPLCalloc(1, nCharCount * 3 + 1));
179 27327 : size_t iDst = 0;
180 27327 : unsigned char *pabyResult =
181 : reinterpret_cast<unsigned char *>(pszResult);
182 636132 : for (size_t i = 0; i < nCharCount; ++i)
183 : {
184 608805 : const unsigned char nChar =
185 608805 : static_cast<unsigned char>(pszSource[i]);
186 608805 : if (nChar <= 127)
187 : {
188 554294 : pszResult[iDst] = pszSource[i];
189 554294 : ++iDst;
190 : }
191 : else
192 : {
193 54511 : const unsigned char nShiftedChar = nChar - 128;
194 54511 : if (convTable[nShiftedChar][0])
195 : {
196 54510 : pabyResult[iDst] = convTable[nShiftedChar][0];
197 54510 : ++iDst;
198 54510 : CPLAssert(convTable[nShiftedChar][1]);
199 54510 : pabyResult[iDst] = convTable[nShiftedChar][1];
200 54510 : ++iDst;
201 54510 : if (convTable[nShiftedChar][2])
202 : {
203 13 : pabyResult[iDst] = convTable[nShiftedChar][2];
204 13 : ++iDst;
205 : }
206 : }
207 : else
208 : {
209 : // Skip the invalid sequence in the input string.
210 1 : if (!bHaveWarned2)
211 : {
212 1 : bHaveWarned2 = true;
213 1 : CPLError(CE_Warning, CPLE_AppDefined,
214 : "One or several characters couldn't be "
215 : "converted correctly from %s to %s. "
216 : "This warning will not be emitted anymore",
217 : pszSrcEncoding, pszDstEncoding);
218 : }
219 : }
220 : }
221 : }
222 :
223 27327 : pszResult[iDst] = 0;
224 27327 : return pszResult;
225 : }
226 : }
227 :
228 : #ifdef _WIN32
229 : const auto MapEncodingToWindowsCodePage = [](const char *pszEncoding)
230 : {
231 : // Cf https://learn.microsoft.com/fr-fr/windows/win32/intl/code-page-identifiers
232 : if (STARTS_WITH(pszEncoding, "CP"))
233 : {
234 : const int nCode = atoi(pszEncoding + strlen("CP"));
235 : if (nCode > 0)
236 : return nCode;
237 : else if (EQUAL(pszEncoding, "CP_OEMCP"))
238 : return CP_OEMCP;
239 : else if (EQUAL(pszEncoding, "CP_ACP"))
240 : return CP_ACP;
241 : }
242 : else if (STARTS_WITH(pszEncoding, "WINDOWS-"))
243 : {
244 : const int nCode = atoi(pszEncoding + strlen("WINDOWS-"));
245 : if (nCode > 0)
246 : return nCode;
247 : }
248 : else if (STARTS_WITH(pszEncoding, "ISO-8859-"))
249 : {
250 : const int nCode = atoi(pszEncoding + strlen("ISO-8859-"));
251 : if ((nCode >= 1 && nCode <= 9) || nCode == 13 || nCode == 15)
252 : return 28590 + nCode;
253 : }
254 :
255 : // Return a negative value, since CP_ACP = 0
256 : return -1;
257 : };
258 :
259 : /* ---------------------------------------------------------------------*/
260 : /* XXX to UTF8 */
261 : /* ---------------------------------------------------------------------*/
262 : if (strcmp(pszDstEncoding, CPL_ENC_UTF8) == 0)
263 : {
264 : const int nCode = MapEncodingToWindowsCodePage(pszSrcEncoding);
265 : if (nCode >= 0)
266 : {
267 : return CPLWin32Recode(pszSource, nCode, CP_UTF8);
268 : }
269 : }
270 :
271 : /* ---------------------------------------------------------------------*/
272 : /* UTF8 to XXX */
273 : /* ---------------------------------------------------------------------*/
274 : if (strcmp(pszSrcEncoding, CPL_ENC_UTF8) == 0)
275 : {
276 : const int nCode = MapEncodingToWindowsCodePage(pszDstEncoding);
277 : if (nCode >= 0)
278 : {
279 : return CPLWin32Recode(pszSource, CP_UTF8, nCode);
280 : }
281 : }
282 : #endif
283 :
284 : /* -------------------------------------------------------------------- */
285 : /* Anything else to UTF-8 is treated as ISO8859-1 to UTF-8 with */
286 : /* a one-time warning. */
287 : /* -------------------------------------------------------------------- */
288 0 : if (strcmp(pszDstEncoding, CPL_ENC_UTF8) == 0)
289 : {
290 0 : const int nCharCount = static_cast<int>(strlen(pszSource));
291 0 : char *pszResult = static_cast<char *>(CPLCalloc(1, nCharCount * 2 + 1));
292 :
293 0 : if (!bHaveWarned1)
294 : {
295 0 : bHaveWarned1 = true;
296 0 : CPLError(CE_Warning, CPLE_AppDefined,
297 : "Recode from %s to UTF-8 not supported, "
298 : "treated as ISO-8859-1 to UTF-8.",
299 : pszSrcEncoding);
300 : }
301 :
302 0 : utf8froma(pszResult, nCharCount * 2 + 1, pszSource, nCharCount);
303 :
304 0 : return pszResult;
305 : }
306 :
307 : /* -------------------------------------------------------------------- */
308 : /* Everything else is treated as a no-op with a warning. */
309 : /* -------------------------------------------------------------------- */
310 : {
311 0 : if (!bHaveWarned3)
312 : {
313 0 : bHaveWarned3 = true;
314 0 : CPLError(CE_Warning, CPLE_AppDefined,
315 : "Recode from %s to %s not supported, no change applied.",
316 : pszSrcEncoding, pszDstEncoding);
317 : }
318 :
319 0 : return CPLStrdup(pszSource);
320 : }
321 : }
322 :
323 : /************************************************************************/
324 : /* CPLRecodeFromWCharStub() */
325 : /************************************************************************/
326 :
327 : /**
328 : * Convert wchar_t string to UTF-8.
329 : *
330 : * Convert a wchar_t string into a multibyte utf-8 string. The only
331 : * guaranteed supported source encoding is CPL_ENC_UCS2, and the only
332 : * guaranteed supported destination encodings are CPL_ENC_UTF8, CPL_ENC_ASCII
333 : * and CPL_ENC_ISO8859_1. In some cases (i.e. using iconv()) other encodings
334 : * may also be supported.
335 : *
336 : * Note that the wchar_t type varies in size on different systems. On
337 : * win32 it is normally 2 bytes, and on unix 4 bytes.
338 : *
339 : * If an error occurs an error may, or may not be posted with CPLError().
340 : *
341 : * @param pwszSource the source wchar_t string, terminated with a 0 wchar_t.
342 : * @param pszSrcEncoding the source encoding, typically CPL_ENC_UCS2.
343 : * @param pszDstEncoding the destination encoding, typically CPL_ENC_UTF8.
344 : *
345 : * @return a zero terminated multi-byte string which should be freed with
346 : * CPLFree(), or NULL if an error occurs.
347 : */
348 :
349 131010 : char *CPLRecodeFromWCharStub(const wchar_t *pwszSource,
350 : const char *pszSrcEncoding,
351 : const char *pszDstEncoding)
352 :
353 : {
354 : /* -------------------------------------------------------------------- */
355 : /* We try to avoid changes of character set. We are just */
356 : /* providing for unicode to unicode. */
357 : /* -------------------------------------------------------------------- */
358 131010 : if (strcmp(pszSrcEncoding, "WCHAR_T") != 0 &&
359 129260 : strcmp(pszSrcEncoding, CPL_ENC_UTF8) != 0 &&
360 129260 : strcmp(pszSrcEncoding, CPL_ENC_UTF16) != 0 &&
361 129260 : strcmp(pszSrcEncoding, CPL_ENC_UCS2) != 0 &&
362 0 : strcmp(pszSrcEncoding, CPL_ENC_UCS4) != 0)
363 : {
364 0 : CPLError(CE_Failure, CPLE_AppDefined,
365 : "Stub recoding implementation does not support "
366 : "CPLRecodeFromWCharStub(...,%s,%s)",
367 : pszSrcEncoding, pszDstEncoding);
368 0 : return nullptr;
369 : }
370 :
371 : /* -------------------------------------------------------------------- */
372 : /* What is the source length. */
373 : /* -------------------------------------------------------------------- */
374 131010 : int nSrcLen = 0;
375 :
376 1933600 : while (pwszSource[nSrcLen] != 0)
377 1802590 : nSrcLen++;
378 :
379 : /* -------------------------------------------------------------------- */
380 : /* Allocate destination buffer plenty big. */
381 : /* -------------------------------------------------------------------- */
382 131010 : const int nDstBufSize = nSrcLen * 4 + 1;
383 : // Nearly worst case.
384 131010 : char *pszResult = static_cast<char *>(CPLMalloc(nDstBufSize));
385 :
386 131010 : if (nSrcLen == 0)
387 : {
388 57959 : pszResult[0] = '\0';
389 57959 : return pszResult;
390 : }
391 :
392 : /* -------------------------------------------------------------------- */
393 : /* Convert, and confirm we had enough space. */
394 : /* -------------------------------------------------------------------- */
395 73051 : const int nDstLen = utf8fromwc(pszResult, nDstBufSize, pwszSource, nSrcLen);
396 73051 : if (nDstLen >= nDstBufSize)
397 : {
398 0 : CPLAssert(false); // too small!
399 : return nullptr;
400 : }
401 :
402 : /* -------------------------------------------------------------------- */
403 : /* If something other than UTF-8 was requested, recode now. */
404 : /* -------------------------------------------------------------------- */
405 73051 : if (strcmp(pszDstEncoding, CPL_ENC_UTF8) == 0)
406 73051 : return pszResult;
407 :
408 : char *pszFinalResult =
409 0 : CPLRecodeStub(pszResult, CPL_ENC_UTF8, pszDstEncoding);
410 :
411 0 : CPLFree(pszResult);
412 :
413 0 : return pszFinalResult;
414 : }
415 :
416 : /************************************************************************/
417 : /* CPLRecodeToWCharStub() */
418 : /************************************************************************/
419 :
420 : /**
421 : * Convert UTF-8 string to a wchar_t string.
422 : *
423 : * Convert a 8bit, multi-byte per character input string into a wide
424 : * character (wchar_t) string. The only guaranteed supported source encodings
425 : * are CPL_ENC_UTF8, CPL_ENC_ASCII and CPL_ENC_ISO8869_1 (LATIN1). The only
426 : * guaranteed supported destination encoding is CPL_ENC_UCS2. Other source
427 : * and destination encodings may be supported depending on the underlying
428 : * implementation.
429 : *
430 : * Note that the wchar_t type varies in size on different systems. On
431 : * win32 it is normally 2 bytes, and on unix 4 bytes.
432 : *
433 : * If an error occurs an error may, or may not be posted with CPLError().
434 : *
435 : * @param pszSource input multi-byte character string.
436 : * @param pszSrcEncoding source encoding, typically CPL_ENC_UTF8.
437 : * @param pszDstEncoding destination encoding, typically CPL_ENC_UCS2.
438 : *
439 : * @return the zero terminated wchar_t string (to be freed with CPLFree()) or
440 : * NULL on error.
441 : *
442 : */
443 :
444 41083 : wchar_t *CPLRecodeToWCharStub(const char *pszSource, const char *pszSrcEncoding,
445 : const char *pszDstEncoding)
446 :
447 : {
448 41083 : char *pszUTF8Source = const_cast<char *>(pszSource);
449 :
450 41083 : if (strcmp(pszSrcEncoding, CPL_ENC_UTF8) != 0 &&
451 0 : strcmp(pszSrcEncoding, CPL_ENC_ASCII) != 0)
452 : {
453 0 : pszUTF8Source = CPLRecodeStub(pszSource, pszSrcEncoding, CPL_ENC_UTF8);
454 0 : if (pszUTF8Source == nullptr)
455 0 : return nullptr;
456 : }
457 :
458 : /* -------------------------------------------------------------------- */
459 : /* We try to avoid changes of character set. We are just */
460 : /* providing for unicode to unicode. */
461 : /* -------------------------------------------------------------------- */
462 41083 : if (strcmp(pszDstEncoding, "WCHAR_T") != 0 &&
463 41083 : strcmp(pszDstEncoding, CPL_ENC_UCS2) != 0 &&
464 0 : strcmp(pszDstEncoding, CPL_ENC_UCS4) != 0 &&
465 0 : strcmp(pszDstEncoding, CPL_ENC_UTF16) != 0)
466 : {
467 0 : CPLError(CE_Failure, CPLE_AppDefined,
468 : "Stub recoding implementation does not support "
469 : "CPLRecodeToWCharStub(...,%s,%s)",
470 : pszSrcEncoding, pszDstEncoding);
471 0 : if (pszUTF8Source != pszSource)
472 0 : CPLFree(pszUTF8Source);
473 0 : return nullptr;
474 : }
475 :
476 : /* -------------------------------------------------------------------- */
477 : /* Do the UTF-8 to UCS-2 recoding. */
478 : /* -------------------------------------------------------------------- */
479 41083 : int nSrcLen = static_cast<int>(strlen(pszUTF8Source));
480 : wchar_t *pwszResult =
481 41083 : static_cast<wchar_t *>(CPLCalloc(sizeof(wchar_t), nSrcLen + 1));
482 :
483 41083 : utf8towc(pszUTF8Source, nSrcLen, pwszResult, nSrcLen + 1);
484 :
485 41083 : if (pszUTF8Source != pszSource)
486 0 : CPLFree(pszUTF8Source);
487 :
488 41083 : return pwszResult;
489 : }
490 :
491 : /************************************************************************/
492 : /* CPLIsUTF8() */
493 : /************************************************************************/
494 :
495 : /**
496 : * Test if a string is encoded as UTF-8.
497 : *
498 : * @param pabyData input string to test
499 : * @param nLen length of the input string, or -1 if the function must compute
500 : * the string length. In which case it must be null terminated.
501 : * @return TRUE if the string is encoded as UTF-8. FALSE otherwise
502 : *
503 : */
504 20825 : int CPLIsUTF8(const char *pabyData, int nLen)
505 : {
506 20825 : if (nLen < 0)
507 14899 : nLen = static_cast<int>(strlen(pabyData));
508 20825 : return utf8test(pabyData, static_cast<unsigned>(nLen)) != 0;
509 : }
510 :
511 : /************************************************************************/
512 : /* ==================================================================== */
513 : /* UTF.C code from FLTK with some modifications. */
514 : /* ==================================================================== */
515 : /************************************************************************/
516 :
517 : /* Set to 1 to turn bad UTF8 bytes into ISO-8859-1. If this is to zero
518 : they are instead turned into the Unicode REPLACEMENT CHARACTER, of
519 : value 0xfffd.
520 : If this is on utf8decode will correctly map most (perhaps all)
521 : human-readable text that is in ISO-8859-1. This may allow you
522 : to completely ignore character sets in your code because virtually
523 : everything is either ISO-8859-1 or UTF-8.
524 : */
525 : #define ERRORS_TO_ISO8859_1 1
526 :
527 : /* Set to 1 to turn bad UTF8 bytes in the 0x80-0x9f range into the
528 : Unicode index for Microsoft's CP1252 character set. You should
529 : also set ERRORS_TO_ISO8859_1. With this a huge amount of more
530 : available text (such as all web pages) are correctly converted
531 : to Unicode.
532 : */
533 : #define ERRORS_TO_CP1252 1
534 :
535 : /* A number of Unicode code points are in fact illegal and should not
536 : be produced by a UTF-8 converter. Turn this on will replace the
537 : bytes in those encodings with errors. If you do this then converting
538 : arbitrary 16-bit data to UTF-8 and then back is not an identity,
539 : which will probably break a lot of software.
540 : */
541 : #define STRICT_RFC3629 0
542 :
543 : #if ERRORS_TO_CP1252
544 : // Codes 0x80..0x9f from the Microsoft CP1252 character set, translated
545 : // to Unicode:
546 : constexpr unsigned short cp1252[32] = {
547 : 0x20ac, 0x0081, 0x201a, 0x0192, 0x201e, 0x2026, 0x2020, 0x2021,
548 : 0x02c6, 0x2030, 0x0160, 0x2039, 0x0152, 0x008d, 0x017d, 0x008f,
549 : 0x0090, 0x2018, 0x2019, 0x201c, 0x201d, 0x2022, 0x2013, 0x2014,
550 : 0x02dc, 0x2122, 0x0161, 0x203a, 0x0153, 0x009d, 0x017e, 0x0178};
551 : #endif
552 :
553 : /************************************************************************/
554 : /* utf8decode() */
555 : /************************************************************************/
556 :
557 : /*
558 : Decode a single UTF-8 encoded character starting at \e p. The
559 : resulting Unicode value (in the range 0-0x10ffff) is returned,
560 : and \e len is set the number of bytes in the UTF-8 encoding
561 : (adding \e len to \e p will point at the next character).
562 :
563 : If \a p points at an illegal UTF-8 encoding, including one that
564 : would go past \e end, or where a code is uses more bytes than
565 : necessary, then *reinterpret_cast<const unsigned char*>(p) is translated as
566 : though it is in the Microsoft CP1252 character set and \e len is set to 1.
567 : Treating errors this way allows this to decode almost any
568 : ISO-8859-1 or CP1252 text that has been mistakenly placed where
569 : UTF-8 is expected, and has proven very useful.
570 :
571 : If you want errors to be converted to error characters (as the
572 : standards recommend), adding a test to see if the length is
573 : unexpectedly 1 will work:
574 :
575 : \code
576 : if( *p & 0x80 )
577 : { // What should be a multibyte encoding.
578 : code = utf8decode(p, end, &len);
579 : if( len<2 ) code = 0xFFFD; // Turn errors into REPLACEMENT CHARACTER.
580 : }
581 : else
582 : { // Handle the 1-byte utf8 encoding:
583 : code = *p;
584 : len = 1;
585 : }
586 : \endcode
587 :
588 : Direct testing for the 1-byte case (as shown above) will also
589 : speed up the scanning of strings where the majority of characters
590 : are ASCII.
591 : */
592 2617 : static unsigned utf8decode(const char *p, const char *end, int *len)
593 : {
594 2617 : unsigned char c = *reinterpret_cast<const unsigned char *>(p);
595 2617 : if (c < 0x80)
596 : {
597 0 : *len = 1;
598 0 : return c;
599 : #if ERRORS_TO_CP1252
600 : }
601 2617 : else if (c < 0xa0)
602 : {
603 40 : *len = 1;
604 40 : return cp1252[c - 0x80];
605 : #endif
606 : }
607 2577 : else if (c < 0xc2)
608 : {
609 10 : goto FAIL;
610 : }
611 2567 : if (p + 1 >= end || (p[1] & 0xc0) != 0x80)
612 71 : goto FAIL;
613 2496 : if (c < 0xe0)
614 : {
615 2488 : *len = 2;
616 2488 : return ((p[0] & 0x1f) << 6) + ((p[1] & 0x3f));
617 : }
618 8 : else if (c == 0xe0)
619 : {
620 0 : if ((reinterpret_cast<const unsigned char *>(p))[1] < 0xa0)
621 0 : goto FAIL;
622 0 : goto UTF8_3;
623 : #if STRICT_RFC3629
624 : }
625 : else if (c == 0xed)
626 : {
627 : // RFC 3629 says surrogate chars are illegal.
628 : if ((reinterpret_cast<const unsigned char *>(p))[1] >= 0xa0)
629 : goto FAIL;
630 : goto UTF8_3;
631 : }
632 : else if (c == 0xef)
633 : {
634 : // 0xfffe and 0xffff are also illegal characters.
635 : if ((reinterpret_cast<const unsigned char *>(p))[1] == 0xbf &&
636 : (reinterpret_cast<const unsigned char *>(p))[2] >= 0xbe)
637 : goto FAIL;
638 : goto UTF8_3;
639 : #endif
640 : }
641 8 : else if (c < 0xf0)
642 : {
643 4 : UTF8_3:
644 4 : if (p + 2 >= end || (p[2] & 0xc0) != 0x80)
645 0 : goto FAIL;
646 4 : *len = 3;
647 4 : return ((p[0] & 0x0f) << 12) + ((p[1] & 0x3f) << 6) + ((p[2] & 0x3f));
648 : }
649 4 : else if (c == 0xf0)
650 : {
651 4 : if ((reinterpret_cast<const unsigned char *>(p))[1] < 0x90)
652 0 : goto FAIL;
653 4 : goto UTF8_4;
654 : }
655 0 : else if (c < 0xf4)
656 : {
657 0 : UTF8_4:
658 4 : if (p + 3 >= end || (p[2] & 0xc0) != 0x80 || (p[3] & 0xc0) != 0x80)
659 0 : goto FAIL;
660 4 : *len = 4;
661 : #if STRICT_RFC3629
662 : // RFC 3629 says all codes ending in fffe or ffff are illegal:
663 : if ((p[1] & 0xf) == 0xf &&
664 : (reinterpret_cast<const unsigned char *>(p))[2] == 0xbf &&
665 : (reinterpret_cast<const unsigned char *>(p))[3] >= 0xbe)
666 : goto FAIL;
667 : #endif
668 4 : return ((p[0] & 0x07) << 18) + ((p[1] & 0x3f) << 12) +
669 4 : ((p[2] & 0x3f) << 6) + ((p[3] & 0x3f));
670 : }
671 0 : else if (c == 0xf4)
672 : {
673 0 : if ((reinterpret_cast<const unsigned char *>(p))[1] > 0x8f)
674 0 : goto FAIL; // After 0x10ffff.
675 0 : goto UTF8_4;
676 : }
677 : else
678 : {
679 0 : FAIL:
680 81 : *len = 1;
681 : #if ERRORS_TO_ISO8859_1
682 81 : return c;
683 : #else
684 : return 0xfffd; // Unicode REPLACEMENT CHARACTER
685 : #endif
686 : }
687 : }
688 :
689 : /************************************************************************/
690 : /* utf8towc() */
691 : /************************************************************************/
692 :
693 : /* Convert a UTF-8 sequence into an array of wchar_t. These
694 : are used by some system calls, especially on Windows.
695 :
696 : \a src points at the UTF-8, and \a srclen is the number of bytes to
697 : convert.
698 :
699 : \a dst points at an array to write, and \a dstlen is the number of
700 : locations in this array. At most \a dstlen-1 words will be
701 : written there, plus a 0 terminating word. Thus this function
702 : will never overwrite the buffer and will always return a
703 : zero-terminated string. If \a dstlen is zero then \a dst can be
704 : null and no data is written, but the length is returned.
705 :
706 : The return value is the number of words that \e would be written
707 : to \a dst if it were long enough, not counting the terminating
708 : zero. If the return value is greater or equal to \a dstlen it
709 : indicates truncation, you can then allocate a new array of size
710 : return+1 and call this again.
711 :
712 : Errors in the UTF-8 are converted as though each byte in the
713 : erroneous string is in the Microsoft CP1252 encoding. This allows
714 : ISO-8859-1 text mistakenly identified as UTF-8 to be printed
715 : correctly.
716 :
717 : Notice that sizeof(wchar_t) is 2 on Windows and is 4 on Linux
718 : and most other systems. Where wchar_t is 16 bits, Unicode
719 : characters in the range 0x10000 to 0x10ffff are converted to
720 : "surrogate pairs" which take two words each (this is called UTF-16
721 : encoding). If wchar_t is 32 bits this rather nasty problem is
722 : avoided.
723 : */
724 41083 : static unsigned utf8towc(const char *src, unsigned srclen, wchar_t *dst,
725 : unsigned dstlen)
726 : {
727 41083 : const char *p = src;
728 41083 : const char *e = src + srclen;
729 41083 : unsigned count = 0;
730 41083 : if (dstlen)
731 : while (true)
732 : {
733 300352 : if (p >= e)
734 : {
735 41083 : dst[count] = 0;
736 41083 : return count;
737 : }
738 259269 : if (!(*p & 0x80))
739 : {
740 : // ASCII
741 259067 : dst[count] = *p++;
742 : }
743 : else
744 : {
745 202 : int len = 0;
746 202 : unsigned ucs = utf8decode(p, e, &len);
747 202 : p += len;
748 : #ifdef _WIN32
749 : if (ucs < 0x10000)
750 : {
751 : dst[count] = static_cast<wchar_t>(ucs);
752 : }
753 : else
754 : {
755 : // Make a surrogate pair:
756 : if (count + 2 >= dstlen)
757 : {
758 : dst[count] = 0;
759 : count += 2;
760 : break;
761 : }
762 : dst[count] = static_cast<wchar_t>(
763 : (((ucs - 0x10000u) >> 10) & 0x3ff) | 0xd800);
764 : dst[++count] = static_cast<wchar_t>((ucs & 0x3ff) | 0xdc00);
765 : }
766 : #else
767 202 : dst[count] = static_cast<wchar_t>(ucs);
768 : #endif
769 : }
770 259269 : if (++count == dstlen)
771 : {
772 0 : dst[count - 1] = 0;
773 0 : break;
774 : }
775 259269 : }
776 : // We filled dst, measure the rest:
777 0 : while (p < e)
778 : {
779 0 : if (!(*p & 0x80))
780 : {
781 0 : p++;
782 : }
783 : else
784 : {
785 0 : int len = 0;
786 : #ifdef _WIN32
787 : const unsigned ucs = utf8decode(p, e, &len);
788 : p += len;
789 : if (ucs >= 0x10000)
790 : ++count;
791 : #else
792 0 : utf8decode(p, e, &len);
793 0 : p += len;
794 : #endif
795 : }
796 0 : ++count;
797 : }
798 :
799 0 : return count;
800 : }
801 :
802 : /************************************************************************/
803 : /* utf8toa() */
804 : /************************************************************************/
805 : /* Convert a UTF-8 sequence into an array of 1-byte characters.
806 :
807 : If the UTF-8 decodes to a character greater than 0xff then it is
808 : replaced with '?'.
809 :
810 : Errors in the UTF-8 are converted as individual bytes, same as
811 : utf8decode() does. This allows ISO-8859-1 text mistakenly identified
812 : as UTF-8 to be printed correctly (and possibly CP1252 on Windows).
813 :
814 : \a src points at the UTF-8, and \a srclen is the number of bytes to
815 : convert.
816 :
817 : Up to \a dstlen bytes are written to \a dst, including a null
818 : terminator. The return value is the number of bytes that would be
819 : written, not counting the null terminator. If greater or equal to
820 : \a dstlen then if you malloc a new array of size n+1 you will have
821 : the space needed for the entire string. If \a dstlen is zero then
822 : nothing is written and this call just measures the storage space
823 : needed.
824 : */
825 49447 : static unsigned int utf8toa(const char *src, unsigned srclen, char *dst,
826 : unsigned dstlen)
827 : {
828 49447 : const char *p = src;
829 49447 : const char *e = src + srclen;
830 49447 : unsigned int count = 0;
831 49447 : if (dstlen)
832 : while (true)
833 : {
834 179832 : if (p >= e)
835 : {
836 49447 : dst[count] = 0;
837 49447 : return count;
838 : }
839 130385 : unsigned char c = *reinterpret_cast<const unsigned char *>(p);
840 130385 : if (c < 0xC2)
841 : {
842 : // ASCII or bad code.
843 129585 : dst[count] = c;
844 129585 : p++;
845 : }
846 : else
847 : {
848 800 : int len = 0;
849 800 : const unsigned int ucs = utf8decode(p, e, &len);
850 800 : p += len;
851 800 : if (ucs < 0x100)
852 : {
853 796 : dst[count] = static_cast<char>(ucs);
854 : }
855 : else
856 : {
857 4 : if (!bHaveWarned4)
858 : {
859 2 : bHaveWarned4 = true;
860 2 : CPLError(
861 : CE_Warning, CPLE_AppDefined,
862 : "One or several characters couldn't be converted "
863 : "correctly from UTF-8 to ISO-8859-1. "
864 : "This warning will not be emitted anymore.");
865 : }
866 4 : dst[count] = '?';
867 : }
868 : }
869 130385 : if (++count >= dstlen)
870 : {
871 0 : dst[count - 1] = 0;
872 0 : break;
873 : }
874 130385 : }
875 : // We filled dst, measure the rest:
876 0 : while (p < e)
877 : {
878 0 : if (!(*p & 0x80))
879 : {
880 0 : p++;
881 : }
882 : else
883 : {
884 0 : int len = 0;
885 0 : utf8decode(p, e, &len);
886 0 : p += len;
887 : }
888 0 : ++count;
889 : }
890 0 : return count;
891 : }
892 :
893 : /************************************************************************/
894 : /* utf8fromwc() */
895 : /************************************************************************/
896 : /* Turn "wide characters" as returned by some system calls
897 : (especially on Windows) into UTF-8.
898 :
899 : Up to \a dstlen bytes are written to \a dst, including a null
900 : terminator. The return value is the number of bytes that would be
901 : written, not counting the null terminator. If greater or equal to
902 : \a dstlen then if you malloc a new array of size n+1 you will have
903 : the space needed for the entire string. If \a dstlen is zero then
904 : nothing is written and this call just measures the storage space
905 : needed.
906 :
907 : \a srclen is the number of words in \a src to convert. On Windows
908 : this is not necessarily the number of characters, due to there
909 : possibly being "surrogate pairs" in the UTF-16 encoding used.
910 : On Unix wchar_t is 32 bits and each location is a character.
911 :
912 : On Unix if a src word is greater than 0x10ffff then this is an
913 : illegal character according to RFC 3629. These are converted as
914 : though they are 0xFFFD (REPLACEMENT CHARACTER). Characters in the
915 : range 0xd800 to 0xdfff, or ending with 0xfffe or 0xffff are also
916 : illegal according to RFC 3629. However I encode these as though
917 : they are legal, so that utf8towc will return the original data.
918 :
919 : On Windows "surrogate pairs" are converted to a single character
920 : and UTF-8 encoded (as 4 bytes). Mismatched halves of surrogate
921 : pairs are converted as though they are individual characters.
922 : */
923 73051 : static unsigned int utf8fromwc(char *dst, unsigned dstlen, const wchar_t *src,
924 : unsigned srclen)
925 : {
926 73051 : unsigned int i = 0;
927 73051 : unsigned int count = 0;
928 73051 : if (dstlen)
929 : while (true)
930 : {
931 1875640 : if (i >= srclen)
932 : {
933 73051 : dst[count] = 0;
934 73051 : return count;
935 : }
936 1802590 : unsigned int ucs = src[i++];
937 1802590 : if (ucs < 0x80U)
938 : {
939 1795580 : dst[count++] = static_cast<char>(ucs);
940 1795580 : if (count >= dstlen)
941 : {
942 0 : dst[count - 1] = 0;
943 0 : break;
944 : }
945 : }
946 7013 : else if (ucs < 0x800U)
947 : {
948 : // 2 bytes.
949 4370 : if (count + 2 >= dstlen)
950 : {
951 0 : dst[count] = 0;
952 0 : count += 2;
953 0 : break;
954 : }
955 4370 : dst[count++] = 0xc0 | static_cast<char>(ucs >> 6);
956 4370 : dst[count++] = 0x80 | static_cast<char>(ucs & 0x3F);
957 : #ifdef _WIN32
958 : }
959 : else if (ucs >= 0xd800 && ucs <= 0xdbff && i < srclen &&
960 : src[i] >= 0xdc00 && src[i] <= 0xdfff)
961 : {
962 : // Surrogate pair.
963 : unsigned int ucs2 = src[i++];
964 : ucs = 0x10000U + ((ucs & 0x3ff) << 10) + (ucs2 & 0x3ff);
965 : // All surrogate pairs turn into 4-byte utf8.
966 : #else
967 : }
968 2643 : else if (ucs >= 0x10000)
969 : {
970 1 : if (ucs > 0x10ffff)
971 : {
972 1 : ucs = 0xfffd;
973 1 : goto J1;
974 : }
975 : #endif
976 0 : if (count + 4 >= dstlen)
977 : {
978 0 : dst[count] = 0;
979 0 : count += 4;
980 0 : break;
981 : }
982 0 : dst[count++] = 0xf0 | static_cast<char>(ucs >> 18);
983 0 : dst[count++] = 0x80 | static_cast<char>((ucs >> 12) & 0x3F);
984 0 : dst[count++] = 0x80 | static_cast<char>((ucs >> 6) & 0x3F);
985 0 : dst[count++] = 0x80 | static_cast<char>(ucs & 0x3F);
986 : }
987 : else
988 : {
989 : #ifndef _WIN32
990 2642 : J1:
991 : #endif
992 : // All others are 3 bytes:
993 2643 : if (count + 3 >= dstlen)
994 : {
995 0 : dst[count] = 0;
996 0 : count += 3;
997 0 : break;
998 : }
999 2643 : dst[count++] = 0xe0 | static_cast<char>(ucs >> 12);
1000 2643 : dst[count++] = 0x80 | static_cast<char>((ucs >> 6) & 0x3F);
1001 2643 : dst[count++] = 0x80 | static_cast<char>(ucs & 0x3F);
1002 : }
1003 1802590 : }
1004 :
1005 : // We filled dst, measure the rest:
1006 0 : while (i < srclen)
1007 : {
1008 0 : unsigned int ucs = src[i++];
1009 0 : if (ucs < 0x80U)
1010 : {
1011 0 : count++;
1012 : }
1013 0 : else if (ucs < 0x800U)
1014 : {
1015 : // 2 bytes.
1016 0 : count += 2;
1017 : #ifdef _WIN32
1018 : }
1019 : else if (ucs >= 0xd800 && ucs <= 0xdbff && i < srclen - 1 &&
1020 : src[i + 1] >= 0xdc00 && src[i + 1] <= 0xdfff)
1021 : {
1022 : // Surrogate pair.
1023 : ++i;
1024 : #else
1025 : }
1026 0 : else if (ucs >= 0x10000 && ucs <= 0x10ffff)
1027 : {
1028 : #endif
1029 0 : count += 4;
1030 : }
1031 : else
1032 : {
1033 0 : count += 3;
1034 : }
1035 : }
1036 0 : return count;
1037 : }
1038 :
1039 : /************************************************************************/
1040 : /* utf8froma() */
1041 : /************************************************************************/
1042 :
1043 : /* Convert an ISO-8859-1 (i.e. normal c-string) byte stream to UTF-8.
1044 :
1045 : It is possible this should convert Microsoft's CP1252 to UTF-8
1046 : instead. This would translate the codes in the range 0x80-0x9f
1047 : to different characters. Currently it does not do this.
1048 :
1049 : Up to \a dstlen bytes are written to \a dst, including a null
1050 : terminator. The return value is the number of bytes that would be
1051 : written, not counting the null terminator. If greater or equal to
1052 : \a dstlen then if you malloc a new array of size n+1 you will have
1053 : the space needed for the entire string. If \a dstlen is zero then
1054 : nothing is written and this call just measures the storage space
1055 : needed.
1056 :
1057 : \a srclen is the number of bytes in \a src to convert.
1058 :
1059 : If the return value equals \a srclen then this indicates that
1060 : no conversion is necessary, as only ASCII characters are in the
1061 : string.
1062 : */
1063 1958380 : static unsigned utf8froma(char *dst, unsigned dstlen, const char *src,
1064 : unsigned srclen)
1065 : {
1066 1958380 : const char *p = src;
1067 1958380 : const char *e = src + srclen;
1068 1958380 : unsigned count = 0;
1069 1958380 : if (dstlen)
1070 : while (true)
1071 : {
1072 31407200 : if (p >= e)
1073 : {
1074 1958380 : dst[count] = 0;
1075 1958380 : return count;
1076 : }
1077 29448800 : unsigned char ucs = *reinterpret_cast<const unsigned char *>(p);
1078 29448800 : p++;
1079 29448800 : if (ucs < 0x80U)
1080 : {
1081 29394500 : dst[count++] = ucs;
1082 29394500 : if (count >= dstlen)
1083 : {
1084 0 : dst[count - 1] = 0;
1085 0 : break;
1086 : }
1087 : }
1088 : else
1089 : {
1090 : // 2 bytes (note that CP1252 translate could make 3 bytes!)
1091 54280 : if (count + 2 >= dstlen)
1092 : {
1093 0 : dst[count] = 0;
1094 0 : count += 2;
1095 0 : break;
1096 : }
1097 54280 : dst[count++] = 0xc0 | (ucs >> 6);
1098 54280 : dst[count++] = 0x80 | (ucs & 0x3F);
1099 : }
1100 29448800 : }
1101 :
1102 : // We filled dst, measure the rest:
1103 0 : while (p < e)
1104 : {
1105 0 : unsigned char ucs = *reinterpret_cast<const unsigned char *>(p);
1106 0 : p++;
1107 0 : if (ucs < 0x80U)
1108 : {
1109 0 : count++;
1110 : }
1111 : else
1112 : {
1113 0 : count += 2;
1114 : }
1115 : }
1116 :
1117 0 : return count;
1118 : }
1119 :
1120 : #ifdef _WIN32
1121 :
1122 : /************************************************************************/
1123 : /* CPLWin32Recode() */
1124 : /************************************************************************/
1125 :
1126 : /* Convert an CODEPAGE (i.e. normal c-string) byte stream
1127 : to another CODEPAGE (i.e. normal c-string) byte stream.
1128 :
1129 : \a src is target c-string byte stream (including a null terminator).
1130 : \a src_code_page is target c-string byte code page.
1131 : \a dst_code_page is destination c-string byte code page.
1132 :
1133 : UTF7 65000
1134 : UTF8 65001
1135 : OEM-US 437
1136 : OEM-ALABIC 720
1137 : OEM-GREEK 737
1138 : OEM-BALTIC 775
1139 : OEM-MLATIN1 850
1140 : OEM-LATIN2 852
1141 : OEM-CYRILLIC 855
1142 : OEM-TURKISH 857
1143 : OEM-MLATIN1P 858
1144 : OEM-HEBREW 862
1145 : OEM-RUSSIAN 866
1146 :
1147 : THAI 874
1148 : SJIS 932
1149 : GBK 936
1150 : KOREA 949
1151 : BIG5 950
1152 :
1153 : EUROPE 1250
1154 : CYRILLIC 1251
1155 : LATIN1 1252
1156 : GREEK 1253
1157 : TURKISH 1254
1158 : HEBREW 1255
1159 : ARABIC 1256
1160 : BALTIC 1257
1161 : VIETNAM 1258
1162 :
1163 : ISO-LATIN1 28591
1164 : ISO-LATIN2 28592
1165 : ISO-LATIN3 28593
1166 : ISO-BALTIC 28594
1167 : ISO-CYRILLIC 28595
1168 : ISO-ARABIC 28596
1169 : ISO-HEBREW 28598
1170 : ISO-TURKISH 28599
1171 : ISO-LATIN9 28605
1172 :
1173 : ISO-2022-JP 50220
1174 :
1175 : */
1176 :
1177 : char *CPLWin32Recode(const char *src, unsigned src_code_page,
1178 : unsigned dst_code_page)
1179 : {
1180 : // Convert from source code page to Unicode.
1181 :
1182 : // Compute the length in wide characters.
1183 : int wlen = MultiByteToWideChar(src_code_page, MB_ERR_INVALID_CHARS, src, -1,
1184 : nullptr, 0);
1185 : if (wlen == 0 && GetLastError() == ERROR_NO_UNICODE_TRANSLATION)
1186 : {
1187 : if (!bHaveWarned5)
1188 : {
1189 : bHaveWarned5 = true;
1190 : CPLError(
1191 : CE_Warning, CPLE_AppDefined,
1192 : "One or several characters could not be translated from CP%d. "
1193 : "This warning will not be emitted anymore.",
1194 : src_code_page);
1195 : }
1196 :
1197 : // Retry now without MB_ERR_INVALID_CHARS flag.
1198 : wlen = MultiByteToWideChar(src_code_page, 0, src, -1, nullptr, 0);
1199 : }
1200 :
1201 : // Do the actual conversion.
1202 : wchar_t *tbuf =
1203 : static_cast<wchar_t *>(CPLCalloc(sizeof(wchar_t), wlen + 1));
1204 : tbuf[wlen] = 0;
1205 : MultiByteToWideChar(src_code_page, 0, src, -1, tbuf, wlen + 1);
1206 :
1207 : // Convert from Unicode to destination code page.
1208 :
1209 : // Compute the length in chars.
1210 : BOOL bUsedDefaultChar = FALSE;
1211 : int len = 0;
1212 : if (dst_code_page == CP_UTF7 || dst_code_page == CP_UTF8)
1213 : len = WideCharToMultiByte(dst_code_page, 0, tbuf, -1, nullptr, 0,
1214 : nullptr, nullptr);
1215 : else
1216 : len = WideCharToMultiByte(dst_code_page, 0, tbuf, -1, nullptr, 0,
1217 : nullptr, &bUsedDefaultChar);
1218 : if (bUsedDefaultChar)
1219 : {
1220 : if (!bHaveWarned6)
1221 : {
1222 : bHaveWarned6 = true;
1223 : CPLError(
1224 : CE_Warning, CPLE_AppDefined,
1225 : "One or several characters could not be translated to CP%d. "
1226 : "This warning will not be emitted anymore.",
1227 : dst_code_page);
1228 : }
1229 : }
1230 :
1231 : // Do the actual conversion.
1232 : char *pszResult = static_cast<char *>(CPLCalloc(sizeof(char), len + 1));
1233 : WideCharToMultiByte(dst_code_page, 0, tbuf, -1, pszResult, len + 1, nullptr,
1234 : nullptr);
1235 : pszResult[len] = 0;
1236 :
1237 : CPLFree(tbuf);
1238 :
1239 : return pszResult;
1240 : }
1241 :
1242 : #endif
1243 :
1244 : /*
1245 : ** For now we disable the rest which is locale() related. We may need
1246 : ** parts of it later.
1247 : */
1248 :
1249 : #ifdef notdef
1250 :
1251 : #ifdef _WIN32
1252 : #include <windows.h>
1253 : #endif
1254 :
1255 : /*! Return true if the "locale" seems to indicate that UTF-8 encoding
1256 : is used. If true the utf8tomb and utf8frommb don't do anything
1257 : useful.
1258 :
1259 : <i>It is highly recommended that you change your system so this
1260 : does return true.</i> On Windows this is done by setting the
1261 : "codepage" to CP_UTF8. On Unix this is done by setting $LC_CTYPE
1262 : to a string containing the letters "utf" or "UTF" in it, or by
1263 : deleting all $LC* and $LANG environment variables. In the future
1264 : it is likely that all non-Asian Unix systems will return true,
1265 : due to the compatibility of UTF-8 with ISO-8859-1.
1266 : */
1267 : int utf8locale(void)
1268 : {
1269 : static int ret = 2;
1270 : if (ret == 2)
1271 : {
1272 : #ifdef _WIN32
1273 : ret = GetACP() == CP_UTF8;
1274 : #else
1275 : char *s;
1276 : ret = 1; // assume UTF-8 if no locale
1277 : if (((s = getenv("LC_CTYPE")) && *s) ||
1278 : ((s = getenv("LC_ALL")) && *s) || ((s = getenv("LANG")) && *s))
1279 : {
1280 : ret = strstr(s, "utf") || strstr(s, "UTF");
1281 : }
1282 : #endif
1283 : }
1284 :
1285 : return ret;
1286 : }
1287 :
1288 : /*! Convert the UTF-8 used by FLTK to the locale-specific encoding
1289 : used for filenames (and sometimes used for data in files).
1290 : Unfortunately due to stupid design you will have to do this as
1291 : needed for filenames. This is a bug on both Unix and Windows.
1292 :
1293 : Up to \a dstlen bytes are written to \a dst, including a null
1294 : terminator. The return value is the number of bytes that would be
1295 : written, not counting the null terminator. If greater or equal to
1296 : \a dstlen then if you malloc a new array of size n+1 you will have
1297 : the space needed for the entire string. If \a dstlen is zero then
1298 : nothing is written and this call just measures the storage space
1299 : needed.
1300 :
1301 : If utf8locale() returns true then this does not change the data.
1302 : It is copied and truncated as necessary to
1303 : the destination buffer and \a srclen is always returned. */
1304 : unsigned utf8tomb(const char *src, unsigned srclen, char *dst, unsigned dstlen)
1305 : {
1306 : if (!utf8locale())
1307 : {
1308 : #ifdef _WIN32
1309 : wchar_t lbuf[1024] = {};
1310 : wchar_t *buf = lbuf;
1311 : unsigned length = utf8towc(src, srclen, buf, 1024);
1312 : unsigned ret;
1313 : if (length >= 1024)
1314 : {
1315 : buf =
1316 : static_cast<wchar_t *>(malloc((length + 1) * sizeof(wchar_t)));
1317 : utf8towc(src, srclen, buf, length + 1);
1318 : }
1319 : if (dstlen)
1320 : {
1321 : // apparently this does not null-terminate, even though msdn
1322 : // documentation claims it does:
1323 : ret = WideCharToMultiByte(GetACP(), 0, buf, length, dst, dstlen, 0,
1324 : 0);
1325 : dst[ret] = 0;
1326 : }
1327 : // if it overflows or measuring length, get the actual length:
1328 : if (dstlen == 0 || ret >= dstlen - 1)
1329 : ret = WideCharToMultiByte(GetACP(), 0, buf, length, 0, 0, 0, 0);
1330 : if (buf != lbuf)
1331 : free((void *)buf);
1332 : return ret;
1333 : #else
1334 : wchar_t lbuf[1024] = {};
1335 : wchar_t *buf = lbuf;
1336 : unsigned length = utf8towc(src, srclen, buf, 1024);
1337 : if (length >= 1024)
1338 : {
1339 : buf =
1340 : static_cast<wchar_t *>(malloc((length + 1) * sizeof(wchar_t)));
1341 : utf8towc(src, srclen, buf, length + 1);
1342 : }
1343 : int ret = 0;
1344 : if (dstlen)
1345 : {
1346 : ret = wcstombs(dst, buf, dstlen);
1347 : if (ret >= dstlen - 1)
1348 : ret = wcstombs(0, buf, 0);
1349 : }
1350 : else
1351 : {
1352 : ret = wcstombs(0, buf, 0);
1353 : }
1354 : if (buf != lbuf)
1355 : free((void *)buf);
1356 : if (ret >= 0)
1357 : return (unsigned)ret;
1358 : // On any errors we return the UTF-8 as raw text...
1359 : #endif
1360 : }
1361 : // Identity transform:
1362 : if (srclen < dstlen)
1363 : {
1364 : memcpy(dst, src, srclen);
1365 : dst[srclen] = 0;
1366 : }
1367 : else
1368 : {
1369 : memcpy(dst, src, dstlen - 1);
1370 : dst[dstlen - 1] = 0;
1371 : }
1372 : return srclen;
1373 : }
1374 :
1375 : /*! Convert a filename from the locale-specific multibyte encoding
1376 : used by Windows to UTF-8 as used by FLTK.
1377 :
1378 : Up to \a dstlen bytes are written to \a dst, including a null
1379 : terminator. The return value is the number of bytes that would be
1380 : written, not counting the null terminator. If greater or equal to
1381 : \a dstlen then if you malloc a new array of size n+1 you will have
1382 : the space needed for the entire string. If \a dstlen is zero then
1383 : nothing is written and this call just measures the storage space
1384 : needed.
1385 :
1386 : On Unix or on Windows when a UTF-8 locale is in effect, this
1387 : does not change the data. It is copied and truncated as necessary to
1388 : the destination buffer and \a srclen is always returned.
1389 : You may also want to check if utf8test() returns non-zero, so that
1390 : the filesystem can store filenames in UTF-8 encoding regardless of
1391 : the locale.
1392 : */
1393 : unsigned utf8frommb(char *dst, unsigned dstlen, const char *src,
1394 : unsigned srclen)
1395 : {
1396 : if (!utf8locale())
1397 : {
1398 : #ifdef _WIN32
1399 : wchar_t lbuf[1024] = {};
1400 : wchar_t *buf = lbuf;
1401 : unsigned ret;
1402 : const unsigned length =
1403 : MultiByteToWideChar(GetACP(), 0, src, srclen, buf, 1024);
1404 : if (length >= 1024)
1405 : {
1406 : length = MultiByteToWideChar(GetACP(), 0, src, srclen, 0, 0);
1407 : buf = static_cast<wchar_t *>(malloc(length * sizeof(wchar_t)));
1408 : MultiByteToWideChar(GetACP(), 0, src, srclen, buf, length);
1409 : }
1410 : ret = utf8fromwc(dst, dstlen, buf, length);
1411 : if (buf != lbuf)
1412 : free(buf);
1413 : return ret;
1414 : #else
1415 : wchar_t lbuf[1024] = {};
1416 : wchar_t *buf = lbuf;
1417 : const int length = mbstowcs(buf, src, 1024);
1418 : if (length >= 1024)
1419 : {
1420 : length = mbstowcs(0, src, 0) + 1;
1421 : buf =
1422 : static_cast<wchar_t *>(malloc(length * sizeof(unsigned short)));
1423 : mbstowcs(buf, src, length);
1424 : }
1425 : if (length >= 0)
1426 : {
1427 : const unsigned ret = utf8fromwc(dst, dstlen, buf, length);
1428 : if (buf != lbuf)
1429 : free(buf);
1430 : return ret;
1431 : }
1432 : // Errors in conversion return the UTF-8 unchanged.
1433 : #endif
1434 : }
1435 : // Identity transform:
1436 : if (srclen < dstlen)
1437 : {
1438 : memcpy(dst, src, srclen);
1439 : dst[srclen] = 0;
1440 : }
1441 : else
1442 : {
1443 : memcpy(dst, src, dstlen - 1);
1444 : dst[dstlen - 1] = 0;
1445 : }
1446 : return srclen;
1447 : }
1448 :
1449 : #endif // def notdef - disabled locale specific stuff.
1450 :
1451 : /*! Examines the first \a srclen bytes in \a src and return a verdict
1452 : on whether it is UTF-8 or not.
1453 : - Returns 0 if there is any illegal UTF-8 sequences, using the
1454 : same rules as utf8decode(). Note that some UCS values considered
1455 : illegal by RFC 3629, such as 0xffff, are considered legal by this.
1456 : - Returns 1 if there are only single-byte characters (i.e. no bytes
1457 : have the high bit set). This is legal UTF-8, but also indicates
1458 : plain ASCII. It also returns 1 if \a srclen is zero.
1459 : - Returns 2 if there are only characters less than 0x800.
1460 : - Returns 3 if there are only characters less than 0x10000.
1461 : - Returns 4 if there are characters in the 0x10000 to 0x10ffff range.
1462 :
1463 : Because there are many illegal sequences in UTF-8, it is almost
1464 : impossible for a string in another encoding to be confused with
1465 : UTF-8. This is very useful for transitioning Unix to UTF-8
1466 : filenames, you can simply test each filename with this to decide
1467 : if it is UTF-8 or in the locale encoding. My hope is that if
1468 : this is done we will be able to cleanly transition to a locale-less
1469 : encoding.
1470 : */
1471 :
1472 20825 : static int utf8test(const char *src, unsigned srclen)
1473 : {
1474 20825 : int ret = 1;
1475 20825 : const char *p = src;
1476 20825 : const char *e = src + srclen;
1477 1785900 : while (p < e)
1478 : {
1479 1765130 : if (*p == 0)
1480 0 : return 0;
1481 1765130 : if (*p & 0x80)
1482 : {
1483 1615 : int len = 0;
1484 1615 : utf8decode(p, e, &len);
1485 1615 : if (len < 2)
1486 55 : return 0;
1487 1560 : if (len > ret)
1488 555 : ret = len;
1489 1560 : p += len;
1490 : }
1491 : else
1492 : {
1493 1763510 : p++;
1494 : }
1495 : }
1496 20770 : return ret;
1497 : }
|