1 /*****************************************************************************
2 * unicode.c: Unicode <-> locale functions
3 *****************************************************************************
4 * Copyright (C) 2005-2006 VLC authors and VideoLAN
5 * Copyright © 2005-2010 Rémi Denis-Courmont
7 * Authors: Rémi Denis-Courmont <rem # videolan.org>
9 * This program is free software; you can redistribute it and/or modify it
10 * under the terms of the GNU Lesser General Public License as published by
11 * the Free Software Foundation; either version 2.1 of the License, or
12 * (at your option) any later version.
14 * This program is distributed in the hope that it will be useful,
15 * but WITHOUT ANY WARRANTY; without even the implied warranty of
16 * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
17 * GNU Lesser General Public License for more details.
19 * You should have received a copy of the GNU Lesser General Public License
20 * along with this program; if not, write to the Free Software Foundation,
21 * Inc., 51 Franklin Street, Fifth Floor, Boston MA 02110-1301, USA.
22 *****************************************************************************/
24 /*****************************************************************************
26 *****************************************************************************/
31 #include <vlc_common.h>
34 #include <vlc_charset.h>
41 #include <sys/types.h>
51 * Formats an UTF-8 string as vfprintf(), then print it, with
52 * appropriate conversion to local encoding.
54 int utf8_vfprintf( FILE *stream, const char *fmt, va_list ap )
57 return vfprintf (stream, fmt, ap);
60 int res = vasprintf (&str, fmt, ap);
61 if (unlikely(res == -1))
65 /* Writing to the console is a lot of fun on Microsoft Windows.
66 * If you use the standard I/O functions, you must use the OEM code page,
67 * which is different from the usual ANSI code page. Or maybe not, if the
68 * user called "chcp". Anyway, we prefer Unicode. */
69 int fd = _fileno (stream);
70 if (likely(fd != -1) && _isatty (fd))
72 wchar_t *wide = ToWide (str);
73 if (likely(wide != NULL))
75 HANDLE h = (HANDLE)((uintptr_t)_get_osfhandle (fd));
77 /* XXX: It is not clear whether WriteConsole() wants the number of
78 * Unicode characters or the size of the wchar_t array. */
79 BOOL ok = WriteConsoleW (h, wide, wcslen (wide), &out, NULL);
86 char *ansi = ToANSI (str);
101 * Formats an UTF-8 string as fprintf(), then print it, with
102 * appropriate conversion to local encoding.
104 int utf8_fprintf( FILE *stream, const char *fmt, ... )
110 res = utf8_vfprintf( stream, fmt, ap );
117 * Converts the first character from a UTF-8 sequence into a code point.
119 * @param str an UTF-8 bytes sequence
120 * @return 0 if str points to an empty string, i.e. the first character is NUL;
121 * number of bytes that the first character occupies (from 1 to 4) otherwise;
122 * -1 if the byte sequence was not a valid UTF-8 sequence.
124 size_t vlc_towc (const char *str, uint32_t *restrict pwc)
126 uint8_t *ptr = (uint8_t *)str, c;
129 assert (str != NULL);
132 if (unlikely(c > 0xF4))
135 int charlen = clz8 (c ^ 0xFF);
138 case 0: // 7-bit ASCII character -> short cut
142 case 1: // continuation byte -> error
146 if (unlikely(c < 0xC2)) // ASCII overlong
148 cp = (c & 0x1F) << 6;
152 cp = (c & 0x0F) << 12;
156 cp = (c & 0x07) << 16;
163 /* Unrolled continuation bytes decoding */
168 if (unlikely((c >> 6) != 2)) // not a continuation byte
170 cp |= (c & 0x3f) << 12;
172 if (unlikely(cp >= 0x110000)) // beyond Unicode range
177 if (unlikely((c >> 6) != 2)) // not a continuation byte
179 cp |= (c & 0x3f) << 6;
181 if (unlikely(cp >= 0xD800 && cp < 0xE000)) // UTF-16 surrogate
183 if (unlikely(cp < (1u << (5 * charlen - 4)))) // non-ASCII overlong
188 if (unlikely((c >> 6) != 2)) // not a continuation byte
199 * Look for an UTF-8 string within another one in a case-insensitive fashion.
200 * Beware that this is quite slow. Contrary to strcasestr(), this function
201 * works regardless of the system character encoding, and handles multibyte
202 * code points correctly.
204 * @param haystack string to look into
205 * @param needle string to look for
206 * @return a pointer to the first occurence of the needle within the haystack,
207 * or NULL if no occurence were found.
209 char *vlc_strcasestr (const char *haystack, const char *needle)
215 const char *h = haystack, *n = needle;
221 s = vlc_towc (n, &cpn);
223 return (char *)haystack;
228 s = vlc_towc (h, &cph);
229 if (s <= 0 || towlower (cph) != towlower (cpn))
234 s = vlc_towc (haystack, &(uint32_t) { 0 });
243 * Replaces invalid/overlong UTF-8 sequences with question marks.
244 * Note that it is not possible to convert from Latin-1 to UTF-8 on the fly,
245 * so we don't try that, even though it would be less disruptive.
247 * @return str if it was valid UTF-8, NULL if not.
249 char *EnsureUTF8( char *str )
255 while ((n = vlc_towc (str, &cp)) != 0)
256 if (likely(n != (size_t)-1))
268 * Checks whether a string is a valid UTF-8 byte sequence.
270 * @param str nul-terminated string to be checked
272 * @return str if it was valid UTF-8, NULL if not.
274 const char *IsUTF8( const char *str )
279 while ((n = vlc_towc (str, &cp)) != 0)
280 if (likely(n != (size_t)-1))
288 * Converts a string from the given character encoding to utf-8.
290 * @return a nul-terminated utf-8 string, or null in case of error.
291 * The result must be freed using free().
293 char *FromCharset(const char *charset, const void *data, size_t data_size)
295 vlc_iconv_t handle = vlc_iconv_open ("UTF-8", charset);
296 if (handle == (vlc_iconv_t)(-1))
300 for(unsigned mul = 4; mul < 8; mul++ )
302 size_t in_size = data_size;
303 const char *in = data;
304 size_t out_max = mul * data_size;
305 char *tmp = out = malloc (1 + out_max);
309 if (vlc_iconv (handle, &in, &in_size, &tmp, &out_max) != (size_t)(-1)) {
319 vlc_iconv_close(handle);
324 * Converts a nul-terminated UTF-8 string to a given character encoding.
325 * @param charset iconv name of the character set
326 * @param in nul-terminated UTF-8 string
327 * @param outsize pointer to hold the byte size of result
329 * @return A pointer to the result, which must be released using free().
330 * The UTF-8 nul terminator is included in the conversion if the target
331 * character encoding supports it. However it is not included in the returned
333 * In case of error, NULL is returned and the byte size is undefined.
335 void *ToCharset(const char *charset, const char *in, size_t *outsize)
337 vlc_iconv_t hd = vlc_iconv_open (charset, "UTF-8");
338 if (hd == (vlc_iconv_t)(-1))
341 const size_t inlen = strlen (in);
344 for (unsigned mul = 4; mul < 16; mul++)
346 size_t outlen = mul * (inlen + 1);
347 res = malloc (outlen);
348 if (unlikely(res == NULL))
351 const char *inp = in;
354 size_t outb = outlen - mul;
356 if (vlc_iconv (hd, &inp, &inb, &outp, &outb) != (size_t)(-1))
358 *outsize = outlen - mul - outb;
360 inb = 1; /* append nul terminator if possible */
361 if (vlc_iconv (hd, &inp, &inb, &outp, &outb) != (size_t)(-1))
363 if (errno == EILSEQ) /* cannot translate nul terminator!? */
369 if (errno != E2BIG) /* conversion failure */
372 vlc_iconv_close (hd);