Age Owner Branch data TLA Line data Source code
1 : : /*-----------------------------------------------------------------------
2 : : *
3 : : * PostgreSQL locale utilities
4 : : *
5 : : * Portions Copyright (c) 2002-2026, PostgreSQL Global Development Group
6 : : *
7 : : * src/backend/utils/adt/pg_locale.c
8 : : *
9 : : *-----------------------------------------------------------------------
10 : : */
11 : :
12 : : /*----------
13 : : * Here is how the locale stuff is handled: LC_COLLATE and LC_CTYPE
14 : : * are fixed at CREATE DATABASE time, stored in pg_database, and cannot
15 : : * be changed. Thus, the effects of strcoll(), strxfrm(), isupper(),
16 : : * toupper(), etc. are always in the same fixed locale.
17 : : *
18 : : * LC_MESSAGES is settable at run time and will take effect
19 : : * immediately.
20 : : *
21 : : * The other categories, LC_MONETARY, LC_NUMERIC, and LC_TIME are
22 : : * permanently set to "C", and then we use temporary locale_t
23 : : * objects when we need to look up locale data based on the GUCs
24 : : * of the same name. Information is cached when the GUCs change.
25 : : * The cached information is only used by the formatting functions
26 : : * (to_char, etc.) and the money type. For the user, this should all be
27 : : * transparent.
28 : : *----------
29 : : */
30 : :
31 : :
32 : : #include "postgres.h"
33 : :
34 : : #include <time.h>
35 : : #ifdef USE_ICU
36 : : #include <unicode/ucol.h>
37 : : #endif
38 : :
39 : : #include "access/htup_details.h"
40 : : #include "catalog/pg_collation.h"
41 : : #include "catalog/pg_database.h"
42 : : #include "common/hashfn.h"
43 : : #include "common/string.h"
44 : : #include "mb/pg_wchar.h"
45 : : #include "miscadmin.h"
46 : : #include "utils/builtins.h"
47 : : #include "utils/guc_hooks.h"
48 : : #include "utils/lsyscache.h"
49 : : #include "utils/memutils.h"
50 : : #include "utils/pg_locale.h"
51 : : #include "utils/pg_locale_c.h"
52 : : #include "utils/relcache.h"
53 : : #include "utils/syscache.h"
54 : :
55 : : #ifdef WIN32
56 : : #include <shlwapi.h>
57 : : #endif
58 : :
59 : : /* Error triggered for locale-sensitive subroutines */
60 : : #define PGLOCALE_SUPPORT_ERROR(provider) \
61 : : elog(ERROR, "unsupported collprovider for %s: %c", __func__, provider)
62 : :
63 : : /*
64 : : * This should be large enough that most strings will fit, but small enough
65 : : * that we feel comfortable putting it on the stack
66 : : */
67 : : #define TEXTBUFLEN 1024
68 : :
69 : : #define MAX_L10N_DATA 80
70 : :
71 : : /* pg_locale_builtin.c */
72 : : extern pg_locale_t create_pg_locale_builtin(Oid collid, MemoryContext context);
73 : : extern char *get_collation_actual_version_builtin(const char *collcollate);
74 : :
75 : : /* pg_locale_icu.c */
76 : : #ifdef USE_ICU
77 : : extern UCollator *pg_ucol_open(const char *loc_str);
78 : : extern char *get_collation_actual_version_icu(const char *collcollate);
79 : : #endif
80 : : extern pg_locale_t create_pg_locale_icu(Oid collid, MemoryContext context);
81 : :
82 : : /* pg_locale_libc.c */
83 : : extern pg_locale_t create_pg_locale_libc(Oid collid, MemoryContext context);
84 : : extern char *get_collation_actual_version_libc(const char *collcollate);
85 : :
86 : : /* GUC settings */
87 : : char *locale_messages;
88 : : char *locale_monetary;
89 : : char *locale_numeric;
90 : : char *locale_time;
91 : :
92 : : int icu_validation_level = WARNING;
93 : :
94 : : /*
95 : : * lc_time localization cache.
96 : : *
97 : : * We use only the first 7 or 12 entries of these arrays. The last array
98 : : * element is left as NULL for the convenience of outside code that wants
99 : : * to sequentially scan these arrays.
100 : : */
101 : : char *localized_abbrev_days[7 + 1];
102 : : char *localized_full_days[7 + 1];
103 : : char *localized_abbrev_months[12 + 1];
104 : : char *localized_full_months[12 + 1];
105 : :
106 : : static pg_locale_t default_locale = NULL;
107 : :
108 : : /* indicates whether locale information cache is valid */
109 : : static bool CurrentLocaleConvValid = false;
110 : : static bool CurrentLCTimeValid = false;
111 : :
112 : : static struct pg_locale_struct c_locale = {
113 : : .deterministic = true,
114 : : .collate_is_c = true,
115 : : .ctype_is_c = true,
116 : : };
117 : :
118 : : /* Cache for collation-related knowledge */
119 : :
120 : : typedef struct
121 : : {
122 : : Oid collid; /* hash key: pg_collation OID */
123 : : pg_locale_t locale; /* locale_t struct, or 0 if not valid */
124 : :
125 : : /* needed for simplehash */
126 : : uint32 hash;
127 : : char status;
128 : : } collation_cache_entry;
129 : :
130 : : #define SH_PREFIX collation_cache
131 : : #define SH_ELEMENT_TYPE collation_cache_entry
132 : : #define SH_KEY_TYPE Oid
133 : : #define SH_KEY collid
134 : : #define SH_HASH_KEY(tb, key) murmurhash32((uint32) key)
135 : : #define SH_EQUAL(tb, a, b) (a == b)
136 : : #define SH_GET_HASH(tb, a) a->hash
137 : : #define SH_SCOPE static inline
138 : : #define SH_STORE_HASH
139 : : #define SH_DECLARE
140 : : #define SH_DEFINE
141 : : #include "lib/simplehash.h"
142 : :
143 : : static MemoryContext CollationCacheContext = NULL;
144 : : static collation_cache_hash *CollationCache = NULL;
145 : :
146 : : /*
147 : : * The collation cache is often accessed repeatedly for the same collation, so
148 : : * remember the last one used.
149 : : */
150 : : static Oid last_collation_cache_oid = InvalidOid;
151 : : static pg_locale_t last_collation_cache_locale = NULL;
152 : :
153 : : #if defined(WIN32) && defined(LC_MESSAGES)
154 : : static char *IsoLocaleName(const char *);
155 : : #endif
156 : :
157 : : /*
158 : : * pg_perm_setlocale
159 : : *
160 : : * This wraps the libc function setlocale(), with two additions. First, when
161 : : * changing LC_CTYPE, update gettext's encoding for the current message
162 : : * domain. GNU gettext automatically tracks LC_CTYPE on most platforms, but
163 : : * not on Windows. Second, if the operation is successful, the corresponding
164 : : * LC_XXX environment variable is set to match. By setting the environment
165 : : * variable, we ensure that any subsequent use of setlocale(..., "") will
166 : : * preserve the settings made through this routine. Of course, LC_ALL must
167 : : * also be unset to fully ensure that, but that has to be done elsewhere after
168 : : * all the individual LC_XXX variables have been set correctly. (Thank you
169 : : * Perl for making this kluge necessary.)
170 : : */
171 : : char *
7547 tgl@sss.pgh.pa.us 172 :CBC 42096 : pg_perm_setlocale(int category, const char *locale)
173 : : {
174 : : char *result;
175 : : const char *envvar;
176 : :
177 : : #ifndef WIN32
178 : 42096 : result = setlocale(category, locale);
179 : : #else
180 : :
181 : : /*
182 : : * On Windows, setlocale(LC_MESSAGES) does not work, so just assume that
183 : : * the given value is good and set it in the environment variables. We
184 : : * must ignore attempts to set to "", which means "keep using the old
185 : : * environment value".
186 : : */
187 : : #ifdef LC_MESSAGES
188 : : if (category == LC_MESSAGES)
189 : : {
190 : : result = (char *) locale;
191 : : if (locale == NULL || locale[0] == '\0')
192 : : return result;
193 : : }
194 : : else
195 : : #endif
196 : : result = setlocale(category, locale);
197 : : #endif /* WIN32 */
198 : :
199 [ - + ]: 42096 : if (result == NULL)
7547 tgl@sss.pgh.pa.us 200 :UBC 0 : return result; /* fall out immediately on failure */
201 : :
202 : : /*
203 : : * Use the right encoding in translated messages. Under ENABLE_NLS, let
204 : : * pg_bind_textdomain_codeset() figure it out. Under !ENABLE_NLS, message
205 : : * format strings are ASCII, but database-encoding strings may enter the
206 : : * message via %s. This makes the overall message encoding equal to the
207 : : * database encoding.
208 : : */
4810 noah@leadboat.com 209 [ + + ]:CBC 42096 : if (category == LC_CTYPE)
210 : : {
211 : : static char save_lc_ctype[LOCALE_NAME_BUFLEN];
212 : :
213 : : /* copy setlocale() return value before callee invokes it again */
4086 214 : 19509 : strlcpy(save_lc_ctype, result, sizeof(save_lc_ctype));
215 : 19509 : result = save_lc_ctype;
216 : :
217 : : #ifdef ENABLE_NLS
4810 218 : 19509 : SetMessageEncoding(pg_bind_textdomain_codeset(textdomain(NULL)));
219 : : #else
220 : : SetMessageEncoding(GetDatabaseEncoding());
221 : : #endif
222 : : }
223 : :
7547 tgl@sss.pgh.pa.us 224 [ + + + + : 42096 : switch (category)
+ + - ]
225 : : {
226 : 2175 : case LC_COLLATE:
227 : 2175 : envvar = "LC_COLLATE";
228 : 2175 : break;
229 : 19509 : case LC_CTYPE:
230 : 19509 : envvar = "LC_CTYPE";
231 : 19509 : break;
232 : : #ifdef LC_MESSAGES
233 : 13887 : case LC_MESSAGES:
234 : 13887 : envvar = "LC_MESSAGES";
235 : : #ifdef WIN32
236 : : result = IsoLocaleName(locale);
237 : : if (result == NULL)
238 : : result = (char *) locale;
239 : : elog(DEBUG3, "IsoLocaleName() executed; locale: \"%s\"", result);
240 : : #endif /* WIN32 */
241 : 13887 : break;
242 : : #endif /* LC_MESSAGES */
243 : 2175 : case LC_MONETARY:
244 : 2175 : envvar = "LC_MONETARY";
245 : 2175 : break;
246 : 2175 : case LC_NUMERIC:
247 : 2175 : envvar = "LC_NUMERIC";
248 : 2175 : break;
249 : 2175 : case LC_TIME:
250 : 2175 : envvar = "LC_TIME";
251 : 2175 : break;
7547 tgl@sss.pgh.pa.us 252 :UBC 0 : default:
253 [ # # ]: 0 : elog(FATAL, "unrecognized LC category: %d", category);
254 : : return NULL; /* keep compiler quiet */
255 : : }
256 : :
2066 tgl@sss.pgh.pa.us 257 [ - + ]:CBC 42096 : if (setenv(envvar, result, 1) != 0)
7547 tgl@sss.pgh.pa.us 258 :UBC 0 : return NULL;
259 : :
7547 tgl@sss.pgh.pa.us 260 :CBC 42096 : return result;
261 : : }
262 : :
263 : :
264 : : /*
265 : : * Is the locale name valid for the locale category?
266 : : *
267 : : * If successful, and canonname isn't NULL, a palloc'd copy of the locale's
268 : : * canonical name is stored there. This is especially useful for figuring out
269 : : * what locale name "" means (ie, the server environment value). (Actually,
270 : : * it seems that on most implementations that's the only thing it's good for;
271 : : * we could wish that setlocale gave back a canonically spelled version of
272 : : * the locale name, but typically it doesn't.)
273 : : */
274 : : bool
5268 275 : 44818 : check_locale(int category, const char *locale, char **canonname)
276 : : {
277 : : char *save;
278 : : char *res;
279 : :
280 : : /* Don't let Windows' non-ASCII locale names in. */
691 tmunro@postgresql.or 281 [ - + ]: 44818 : if (!pg_is_ascii(locale))
282 : : {
691 tmunro@postgresql.or 283 [ # # ]:UBC 0 : ereport(WARNING,
284 : : (errcode(ERRCODE_INVALID_PARAMETER_VALUE),
285 : : errmsg("locale name \"%s\" contains non-ASCII characters",
286 : : locale)));
287 : 0 : return false;
288 : : }
289 : :
5268 tgl@sss.pgh.pa.us 290 [ + + ]:CBC 44818 : if (canonname)
291 : 863 : *canonname = NULL; /* in case of failure */
292 : :
6547 heikki.linnakangas@i 293 : 44818 : save = setlocale(category, NULL);
294 [ - + ]: 44818 : if (!save)
6547 heikki.linnakangas@i 295 :UBC 0 : return false; /* won't happen, we hope */
296 : :
297 : : /* save may be pointing at a modifiable scratch variable, see above. */
6547 heikki.linnakangas@i 298 :CBC 44818 : save = pstrdup(save);
299 : :
300 : : /* set the locale with setlocale, to see if it accepts it. */
5268 tgl@sss.pgh.pa.us 301 : 44818 : res = setlocale(category, locale);
302 : :
303 : : /* save canonical name if requested. */
304 [ + + + + ]: 44818 : if (res && canonname)
305 : 861 : *canonname = pstrdup(res);
306 : :
307 : : /* restore old value. */
5474 heikki.linnakangas@i 308 [ - + ]: 44818 : if (!setlocale(category, save))
5268 tgl@sss.pgh.pa.us 309 [ # # ]:UBC 0 : elog(WARNING, "failed to restore old locale \"%s\"", save);
6547 heikki.linnakangas@i 310 :CBC 44818 : pfree(save);
311 : :
312 : : /* Don't let Windows' non-ASCII locale names out. */
691 tmunro@postgresql.or 313 [ + + + + : 44818 : if (canonname && *canonname && !pg_is_ascii(*canonname))
- + ]
314 : : {
691 tmunro@postgresql.or 315 [ # # ]:UBC 0 : ereport(WARNING,
316 : : (errcode(ERRCODE_INVALID_PARAMETER_VALUE),
317 : : errmsg("locale name \"%s\" contains non-ASCII characters",
318 : : *canonname)));
319 : 0 : pfree(*canonname);
320 : 0 : *canonname = NULL;
321 : 0 : return false;
322 : : }
323 : :
5268 tgl@sss.pgh.pa.us 324 :CBC 44818 : return (res != NULL);
325 : : }
326 : :
327 : :
328 : : /*
329 : : * GUC check/assign hooks
330 : : *
331 : : * For most locale categories, the assign hook doesn't actually set the locale
332 : : * permanently, just reset flags so that the next use will cache the
333 : : * appropriate values. (See explanation at the top of this file.)
334 : : *
335 : : * Note: we accept value = "" as selecting the postmaster's environment
336 : : * value, whatever it was (so long as the environment setting is legal).
337 : : * This will have been locked down by an earlier call to pg_perm_setlocale.
338 : : */
339 : : bool
5621 340 : 11801 : check_locale_monetary(char **newval, void **extra, GucSource source)
341 : : {
5268 342 : 11801 : return check_locale(LC_MONETARY, *newval, NULL);
343 : : }
344 : :
345 : : void
5621 346 : 11688 : assign_locale_monetary(const char *newval, void *extra)
347 : : {
348 : 11688 : CurrentLocaleConvValid = false;
349 : 11688 : }
350 : :
351 : : bool
352 : 11805 : check_locale_numeric(char **newval, void **extra, GucSource source)
353 : : {
5268 354 : 11805 : return check_locale(LC_NUMERIC, *newval, NULL);
355 : : }
356 : :
357 : : void
5621 358 : 11696 : assign_locale_numeric(const char *newval, void *extra)
359 : : {
360 : 11696 : CurrentLocaleConvValid = false;
8912 peter_e@gmx.net 361 : 11696 : }
362 : :
363 : : bool
5621 tgl@sss.pgh.pa.us 364 : 11801 : check_locale_time(char **newval, void **extra, GucSource source)
365 : : {
5268 366 : 11801 : return check_locale(LC_TIME, *newval, NULL);
367 : : }
368 : :
369 : : void
5621 370 : 11688 : assign_locale_time(const char *newval, void *extra)
371 : : {
372 : 11688 : CurrentLCTimeValid = false;
373 : 11688 : }
374 : :
375 : : /*
376 : : * We allow LC_MESSAGES to actually be set globally.
377 : : *
378 : : * Note: we normally disallow value = "" because it wouldn't have consistent
379 : : * semantics (it'd effectively just use the previous value). However, this
380 : : * is the value passed for PGC_S_DEFAULT, so don't complain in that case,
381 : : * not even if the attempted setting fails due to invalid environment value.
382 : : * The idea there is just to accept the environment setting *if possible*
383 : : * during startup, until we can read the proper value from postgresql.conf.
384 : : */
385 : : bool
386 : 11834 : check_locale_messages(char **newval, void **extra, GucSource source)
387 : : {
388 [ + + ]: 11834 : if (**newval == '\0')
389 : : {
390 [ + - ]: 3286 : if (source == PGC_S_DEFAULT)
391 : 3286 : return true;
392 : : else
5621 tgl@sss.pgh.pa.us 393 :UBC 0 : return false;
394 : : }
395 : :
396 : : /*
397 : : * LC_MESSAGES category does not exist everywhere, but accept it anyway
398 : : *
399 : : * On Windows, we can't even check the value, so accept blindly
400 : : */
401 : : #if defined(LC_MESSAGES) && !defined(WIN32)
5268 tgl@sss.pgh.pa.us 402 :CBC 8548 : return check_locale(LC_MESSAGES, *newval, NULL);
403 : : #else
404 : : return true;
405 : : #endif
406 : : }
407 : :
408 : : void
5621 409 : 11712 : assign_locale_messages(const char *newval, void *extra)
410 : : {
411 : : /*
412 : : * LC_MESSAGES category does not exist everywhere, but accept it anyway.
413 : : * We ignore failure, as per comment above.
414 : : */
415 : : #ifdef LC_MESSAGES
416 : 11712 : (void) pg_perm_setlocale(LC_MESSAGES, newval);
417 : : #endif
8784 peter_e@gmx.net 418 : 11712 : }
419 : :
420 : :
421 : : /*
422 : : * Frees the malloced content of a struct lconv. (But not the struct
423 : : * itself.) It's important that this not throw elog(ERROR).
424 : : */
425 : : static void
3354 tgl@sss.pgh.pa.us 426 : 4 : free_struct_lconv(struct lconv *s)
427 : : {
1533 peter@eisentraut.org 428 : 4 : free(s->decimal_point);
429 : 4 : free(s->thousands_sep);
430 : 4 : free(s->grouping);
431 : 4 : free(s->int_curr_symbol);
432 : 4 : free(s->currency_symbol);
433 : 4 : free(s->mon_decimal_point);
434 : 4 : free(s->mon_thousands_sep);
435 : 4 : free(s->mon_grouping);
436 : 4 : free(s->positive_sign);
437 : 4 : free(s->negative_sign);
3566 tgl@sss.pgh.pa.us 438 : 4 : }
439 : :
440 : : /*
441 : : * Check that all fields of a struct lconv (or at least, the ones we care
442 : : * about) are non-NULL. The field list must match free_struct_lconv().
443 : : */
444 : : static bool
3354 445 : 35 : struct_lconv_is_valid(struct lconv *s)
446 : : {
3566 447 [ - + ]: 35 : if (s->decimal_point == NULL)
3566 tgl@sss.pgh.pa.us 448 :UBC 0 : return false;
3566 tgl@sss.pgh.pa.us 449 [ - + ]:CBC 35 : if (s->thousands_sep == NULL)
3566 tgl@sss.pgh.pa.us 450 :UBC 0 : return false;
3566 tgl@sss.pgh.pa.us 451 [ - + ]:CBC 35 : if (s->grouping == NULL)
3566 tgl@sss.pgh.pa.us 452 :UBC 0 : return false;
3566 tgl@sss.pgh.pa.us 453 [ - + ]:CBC 35 : if (s->int_curr_symbol == NULL)
3566 tgl@sss.pgh.pa.us 454 :UBC 0 : return false;
3566 tgl@sss.pgh.pa.us 455 [ - + ]:CBC 35 : if (s->currency_symbol == NULL)
3566 tgl@sss.pgh.pa.us 456 :UBC 0 : return false;
3566 tgl@sss.pgh.pa.us 457 [ - + ]:CBC 35 : if (s->mon_decimal_point == NULL)
3566 tgl@sss.pgh.pa.us 458 :UBC 0 : return false;
3566 tgl@sss.pgh.pa.us 459 [ - + ]:CBC 35 : if (s->mon_thousands_sep == NULL)
3566 tgl@sss.pgh.pa.us 460 :UBC 0 : return false;
3566 tgl@sss.pgh.pa.us 461 [ - + ]:CBC 35 : if (s->mon_grouping == NULL)
3566 tgl@sss.pgh.pa.us 462 :UBC 0 : return false;
3566 tgl@sss.pgh.pa.us 463 [ - + ]:CBC 35 : if (s->positive_sign == NULL)
3566 tgl@sss.pgh.pa.us 464 :UBC 0 : return false;
3566 tgl@sss.pgh.pa.us 465 [ - + ]:CBC 35 : if (s->negative_sign == NULL)
3566 tgl@sss.pgh.pa.us 466 :UBC 0 : return false;
3566 tgl@sss.pgh.pa.us 467 :CBC 35 : return true;
468 : : }
469 : :
470 : :
471 : : /*
472 : : * Convert the strdup'd string at *str from the specified encoding to the
473 : : * database encoding.
474 : : */
475 : : static void
476 : 280 : db_encoding_convert(int encoding, char **str)
477 : : {
478 : : char *pstr;
479 : : char *mstr;
480 : :
481 : : /* convert the string to the database encoding */
482 : 280 : pstr = pg_any_to_server(*str, strlen(*str), encoding);
483 [ + - ]: 280 : if (pstr == *str)
484 : 280 : return; /* no conversion happened */
485 : :
486 : : /* need it malloc'd not palloc'd */
5971 itagaki.takahiro@gma 487 :UBC 0 : mstr = strdup(pstr);
3566 tgl@sss.pgh.pa.us 488 [ # # ]: 0 : if (mstr == NULL)
489 [ # # ]: 0 : ereport(ERROR,
490 : : (errcode(ERRCODE_OUT_OF_MEMORY),
491 : : errmsg("out of memory")));
492 : :
493 : : /* replace old string */
494 : 0 : free(*str);
495 : 0 : *str = mstr;
496 : :
497 : 0 : pfree(pstr);
498 : : }
499 : :
500 : :
501 : : /*
502 : : * Return the POSIX lconv struct (contains number/money formatting
503 : : * information) with locale information for all categories.
504 : : */
505 : : struct lconv *
9658 tgl@sss.pgh.pa.us 506 :CBC 1786 : PGLC_localeconv(void)
507 : : {
508 : : static struct lconv CurrentLocaleConv;
509 : : static bool CurrentLocaleConvAllocated = false;
510 : : struct lconv *extlconv;
511 : : struct lconv tmp;
518 peter@eisentraut.org 512 : 1786 : struct lconv worklconv = {0};
513 : :
514 : : /* Did we do it already? */
9098 tgl@sss.pgh.pa.us 515 [ + + ]: 1786 : if (CurrentLocaleConvValid)
516 : 1751 : return &CurrentLocaleConv;
517 : :
518 : : /* Free any already-allocated storage */
3833 519 [ + + ]: 35 : if (CurrentLocaleConvAllocated)
520 : : {
521 : 4 : free_struct_lconv(&CurrentLocaleConv);
522 : 4 : CurrentLocaleConvAllocated = false;
523 : : }
524 : :
525 : : /*
526 : : * Use thread-safe method of obtaining a copy of lconv from the operating
527 : : * system.
528 : : */
518 peter@eisentraut.org 529 [ - + ]: 35 : if (pg_localeconv_r(locale_monetary,
530 : : locale_numeric,
531 : : &tmp) != 0)
518 peter@eisentraut.org 532 [ # # ]:UBC 0 : elog(ERROR,
533 : : "could not get lconv for LC_MONETARY = \"%s\", LC_NUMERIC = \"%s\": %m",
534 : : locale_monetary, locale_numeric);
535 : :
536 : : /* Must copy data now so we can re-encode it. */
518 peter@eisentraut.org 537 :CBC 35 : extlconv = &tmp;
3566 tgl@sss.pgh.pa.us 538 : 35 : worklconv.decimal_point = strdup(extlconv->decimal_point);
539 : 35 : worklconv.thousands_sep = strdup(extlconv->thousands_sep);
540 : 35 : worklconv.grouping = strdup(extlconv->grouping);
541 : 35 : worklconv.int_curr_symbol = strdup(extlconv->int_curr_symbol);
542 : 35 : worklconv.currency_symbol = strdup(extlconv->currency_symbol);
543 : 35 : worklconv.mon_decimal_point = strdup(extlconv->mon_decimal_point);
544 : 35 : worklconv.mon_thousands_sep = strdup(extlconv->mon_thousands_sep);
545 : 35 : worklconv.mon_grouping = strdup(extlconv->mon_grouping);
546 : 35 : worklconv.positive_sign = strdup(extlconv->positive_sign);
547 : 35 : worklconv.negative_sign = strdup(extlconv->negative_sign);
548 : : /* Copy scalar fields as well */
549 : 35 : worklconv.int_frac_digits = extlconv->int_frac_digits;
550 : 35 : worklconv.frac_digits = extlconv->frac_digits;
551 : 35 : worklconv.p_cs_precedes = extlconv->p_cs_precedes;
552 : 35 : worklconv.p_sep_by_space = extlconv->p_sep_by_space;
553 : 35 : worklconv.n_cs_precedes = extlconv->n_cs_precedes;
554 : 35 : worklconv.n_sep_by_space = extlconv->n_sep_by_space;
555 : 35 : worklconv.p_sign_posn = extlconv->p_sign_posn;
556 : 35 : worklconv.n_sign_posn = extlconv->n_sign_posn;
557 : :
558 : : /* Free the contents of the object populated by pg_localeconv_r(). */
518 peter@eisentraut.org 559 : 35 : pg_localeconv_free(&tmp);
560 : :
561 : : /* If any of the preceding strdup calls failed, complain now. */
562 [ - + ]: 35 : if (!struct_lconv_is_valid(&worklconv))
518 peter@eisentraut.org 563 [ # # ]:UBC 0 : ereport(ERROR,
564 : : (errcode(ERRCODE_OUT_OF_MEMORY),
565 : : errmsg("out of memory")));
566 : :
3566 tgl@sss.pgh.pa.us 567 [ + - ]:CBC 35 : PG_TRY();
568 : : {
569 : : int encoding;
570 : :
571 : : /*
572 : : * Now we must perform encoding conversion from whatever's associated
573 : : * with the locales into the database encoding. If we can't identify
574 : : * the encoding implied by LC_NUMERIC or LC_MONETARY (ie we get -1),
575 : : * use PG_SQL_ASCII, which will result in just validating that the
576 : : * strings are OK in the database encoding.
577 : : */
578 : 35 : encoding = pg_get_encoding_from_locale(locale_numeric, true);
2683 579 [ - + ]: 35 : if (encoding < 0)
2683 tgl@sss.pgh.pa.us 580 :UBC 0 : encoding = PG_SQL_ASCII;
581 : :
3566 tgl@sss.pgh.pa.us 582 :CBC 35 : db_encoding_convert(encoding, &worklconv.decimal_point);
583 : 35 : db_encoding_convert(encoding, &worklconv.thousands_sep);
584 : : /* grouping is not text and does not require conversion */
585 : :
586 : 35 : encoding = pg_get_encoding_from_locale(locale_monetary, true);
2683 587 [ - + ]: 35 : if (encoding < 0)
2683 tgl@sss.pgh.pa.us 588 :UBC 0 : encoding = PG_SQL_ASCII;
589 : :
3566 tgl@sss.pgh.pa.us 590 :CBC 35 : db_encoding_convert(encoding, &worklconv.int_curr_symbol);
591 : 35 : db_encoding_convert(encoding, &worklconv.currency_symbol);
592 : 35 : db_encoding_convert(encoding, &worklconv.mon_decimal_point);
593 : 35 : db_encoding_convert(encoding, &worklconv.mon_thousands_sep);
594 : : /* mon_grouping is not text and does not require conversion */
595 : 35 : db_encoding_convert(encoding, &worklconv.positive_sign);
596 : 35 : db_encoding_convert(encoding, &worklconv.negative_sign);
597 : : }
3566 tgl@sss.pgh.pa.us 598 :UBC 0 : PG_CATCH();
599 : : {
600 : 0 : free_struct_lconv(&worklconv);
601 : 0 : PG_RE_THROW();
602 : : }
3566 tgl@sss.pgh.pa.us 603 [ - + ]:CBC 35 : PG_END_TRY();
604 : :
605 : : /*
606 : : * Everything is good, so save the results.
607 : : */
608 : 35 : CurrentLocaleConv = worklconv;
609 : 35 : CurrentLocaleConvAllocated = true;
9098 610 : 35 : CurrentLocaleConvValid = true;
611 : 35 : return &CurrentLocaleConv;
612 : : }
613 : :
614 : : #ifdef WIN32
615 : : /*
616 : : * On Windows, strftime() returns its output in encoding CP_ACP (the default
617 : : * operating system codepage for the computer), which is likely different
618 : : * from SERVER_ENCODING. This is especially important in Japanese versions
619 : : * of Windows which will use SJIS encoding, which we don't support as a
620 : : * server encoding.
621 : : *
622 : : * So, instead of using strftime(), use wcsftime() to return the value in
623 : : * wide characters (internally UTF16) and then convert to UTF8, which we
624 : : * know how to handle directly.
625 : : *
626 : : * Note that this only affects the calls to strftime() in this file, which are
627 : : * used to get the locale-aware strings. Other parts of the backend use
628 : : * pg_strftime(), which isn't locale-aware and does not need to be replaced.
629 : : */
630 : : static size_t
631 : : strftime_l_win32(char *dst, size_t dstlen,
632 : : const char *format, const struct tm *tm, locale_t locale)
633 : : {
634 : : size_t len;
635 : : wchar_t wformat[8]; /* formats used below need 3 chars */
636 : : wchar_t wbuf[MAX_L10N_DATA];
637 : :
638 : : /*
639 : : * Get a wchar_t version of the format string. We only actually use
640 : : * plain-ASCII formats in this file, so we can say that they're UTF8.
641 : : */
642 : : len = MultiByteToWideChar(CP_UTF8, 0, format, -1,
643 : : wformat, lengthof(wformat));
644 : : if (len == 0)
645 : : elog(ERROR, "could not convert format string from UTF-8: error code %lu",
646 : : GetLastError());
647 : :
648 : : len = _wcsftime_l(wbuf, MAX_L10N_DATA, wformat, tm, locale);
649 : : if (len == 0)
650 : : {
651 : : /*
652 : : * wcsftime failed, possibly because the result would not fit in
653 : : * MAX_L10N_DATA. Return 0 with the contents of dst unspecified.
654 : : */
655 : : return 0;
656 : : }
657 : :
658 : : len = WideCharToMultiByte(CP_UTF8, 0, wbuf, len, dst, dstlen - 1,
659 : : NULL, NULL);
660 : : if (len == 0)
661 : : elog(ERROR, "could not convert string to UTF-8: error code %lu",
662 : : GetLastError());
663 : :
664 : : dst[len] = '\0';
665 : :
666 : : return len;
667 : : }
668 : :
669 : : /* redefine strftime_l() */
670 : : #define strftime_l(a,b,c,d,e) strftime_l_win32(a,b,c,d,e)
671 : : #endif /* WIN32 */
672 : :
673 : : /*
674 : : * Subroutine for cache_locale_time().
675 : : * Convert the given string from encoding "encoding" to the database
676 : : * encoding, and store the result at *dst, replacing any previous value.
677 : : */
678 : : static void
2683 679 : 1216 : cache_single_string(char **dst, const char *src, int encoding)
680 : : {
681 : : char *ptr;
682 : : char *olddst;
683 : :
684 : : /* Convert the string to the database encoding, or validate it's OK */
685 : 1216 : ptr = pg_any_to_server(src, strlen(src), encoding);
686 : :
687 : : /* Store the string in long-lived storage, replacing any previous value */
688 : 1216 : olddst = *dst;
689 : 1216 : *dst = MemoryContextStrdup(TopMemoryContext, ptr);
690 [ - + ]: 1216 : if (olddst)
2683 tgl@sss.pgh.pa.us 691 :UBC 0 : pfree(olddst);
692 : :
693 : : /* Might as well clean up any palloc'd conversion result, too */
2683 tgl@sss.pgh.pa.us 694 [ - + ]:CBC 1216 : if (ptr != src)
2683 tgl@sss.pgh.pa.us 695 :UBC 0 : pfree(ptr);
4119 noah@leadboat.com 696 :CBC 1216 : }
697 : :
698 : : /*
699 : : * Update the lc_time localization cache variables if needed.
700 : : */
701 : : void
6674 tgl@sss.pgh.pa.us 702 : 33297 : cache_locale_time(void)
703 : : {
704 : : char buf[(2 * 7 + 2 * 12) * MAX_L10N_DATA];
705 : : char *bufptr;
706 : : time_t timenow;
707 : : struct tm *timeinfo;
708 : : struct tm timeinfobuf;
2683 709 : 33297 : bool strftimefail = false;
710 : : int encoding;
711 : : int i;
712 : : locale_t locale;
713 : :
714 : : /* did we do this already? */
6674 715 [ + + ]: 33297 : if (CurrentLCTimeValid)
716 : 33265 : return;
717 : :
718 [ - + ]: 32 : elog(DEBUG3, "cache_locale_time() executed; locale: \"%s\"", locale_time);
719 : :
517 peter@eisentraut.org 720 : 32 : errno = ENOENT;
721 : : #ifdef WIN32
722 : : locale = _create_locale(LC_ALL, locale_time);
723 : : if (locale == (locale_t) 0)
724 : : _dosmaperr(GetLastError());
725 : : #else
726 : 32 : locale = newlocale(LC_ALL_MASK, locale_time, (locale_t) 0);
727 : : #endif
728 [ - + ]: 32 : if (!locale)
517 peter@eisentraut.org 729 :UBC 0 : report_newlocale_failure(locale_time);
730 : :
731 : : /* We use times close to current time as data for strftime(). */
6674 tgl@sss.pgh.pa.us 732 :CBC 32 : timenow = time(NULL);
734 peter@eisentraut.org 733 : 32 : timeinfo = gmtime_r(&timenow, &timeinfobuf);
734 : :
735 : : /* Store the strftime results in MAX_L10N_DATA-sized portions of buf[] */
2683 tgl@sss.pgh.pa.us 736 : 32 : bufptr = buf;
737 : :
738 : : /*
739 : : * MAX_L10N_DATA is sufficient buffer space for every known locale, and
740 : : * POSIX defines no strftime() errors. (Buffer space exhaustion is not an
741 : : * error.) An implementation might report errors (e.g. ENOMEM) by
742 : : * returning 0 (or, less plausibly, a negative value) and setting errno.
743 : : * Report errno just in case the implementation did that, but clear it in
744 : : * advance of the calls so we don't emit a stale, unrelated errno.
745 : : */
746 : 32 : errno = 0;
747 : :
748 : : /* localized days */
6674 749 [ + + ]: 256 : for (i = 0; i < 7; i++)
750 : : {
751 : 224 : timeinfo->tm_wday = i;
517 peter@eisentraut.org 752 [ - + ]: 224 : if (strftime_l(bufptr, MAX_L10N_DATA, "%a", timeinfo, locale) <= 0)
2683 tgl@sss.pgh.pa.us 753 :UBC 0 : strftimefail = true;
2683 tgl@sss.pgh.pa.us 754 :CBC 224 : bufptr += MAX_L10N_DATA;
517 peter@eisentraut.org 755 [ - + ]: 224 : if (strftime_l(bufptr, MAX_L10N_DATA, "%A", timeinfo, locale) <= 0)
2683 tgl@sss.pgh.pa.us 756 :UBC 0 : strftimefail = true;
2683 tgl@sss.pgh.pa.us 757 :CBC 224 : bufptr += MAX_L10N_DATA;
758 : : }
759 : :
760 : : /* localized months */
6674 761 [ + + ]: 416 : for (i = 0; i < 12; i++)
762 : : {
763 : 384 : timeinfo->tm_mon = i;
764 : 384 : timeinfo->tm_mday = 1; /* make sure we don't have invalid date */
517 peter@eisentraut.org 765 [ - + ]: 384 : if (strftime_l(bufptr, MAX_L10N_DATA, "%b", timeinfo, locale) <= 0)
2683 tgl@sss.pgh.pa.us 766 :UBC 0 : strftimefail = true;
2683 tgl@sss.pgh.pa.us 767 :CBC 384 : bufptr += MAX_L10N_DATA;
517 peter@eisentraut.org 768 [ - + ]: 384 : if (strftime_l(bufptr, MAX_L10N_DATA, "%B", timeinfo, locale) <= 0)
2683 tgl@sss.pgh.pa.us 769 :UBC 0 : strftimefail = true;
2683 tgl@sss.pgh.pa.us 770 :CBC 384 : bufptr += MAX_L10N_DATA;
771 : : }
772 : :
773 : : #ifdef WIN32
774 : : _free_locale(locale);
775 : : #else
517 peter@eisentraut.org 776 : 32 : freelocale(locale);
777 : : #endif
778 : :
779 : : /*
780 : : * At this point we've done our best to clean up, and can throw errors, or
781 : : * call functions that might throw errors, with a clean conscience.
782 : : */
2683 tgl@sss.pgh.pa.us 783 [ - + ]: 32 : if (strftimefail)
517 peter@eisentraut.org 784 [ # # ]:UBC 0 : elog(ERROR, "strftime_l() failed");
785 : :
786 : : #ifndef WIN32
787 : :
788 : : /*
789 : : * As in PGLC_localeconv(), we must convert strftime()'s output from the
790 : : * encoding implied by LC_TIME to the database encoding. If we can't
791 : : * identify the LC_TIME encoding, just perform encoding validation.
792 : : */
2683 tgl@sss.pgh.pa.us 793 :CBC 32 : encoding = pg_get_encoding_from_locale(locale_time, true);
794 [ - + ]: 32 : if (encoding < 0)
2683 tgl@sss.pgh.pa.us 795 :UBC 0 : encoding = PG_SQL_ASCII;
796 : :
797 : : #else
798 : :
799 : : /*
800 : : * On Windows, strftime_win32() always returns UTF8 data, so convert from
801 : : * that if necessary.
802 : : */
803 : : encoding = PG_UTF8;
804 : :
805 : : #endif /* WIN32 */
806 : :
2683 tgl@sss.pgh.pa.us 807 :CBC 32 : bufptr = buf;
808 : :
809 : : /* localized days */
810 [ + + ]: 256 : for (i = 0; i < 7; i++)
811 : : {
812 : 224 : cache_single_string(&localized_abbrev_days[i], bufptr, encoding);
813 : 224 : bufptr += MAX_L10N_DATA;
814 : 224 : cache_single_string(&localized_full_days[i], bufptr, encoding);
815 : 224 : bufptr += MAX_L10N_DATA;
816 : : }
2368 817 : 32 : localized_abbrev_days[7] = NULL;
818 : 32 : localized_full_days[7] = NULL;
819 : :
820 : : /* localized months */
2683 821 [ + + ]: 416 : for (i = 0; i < 12; i++)
822 : : {
823 : 384 : cache_single_string(&localized_abbrev_months[i], bufptr, encoding);
824 : 384 : bufptr += MAX_L10N_DATA;
825 : 384 : cache_single_string(&localized_full_months[i], bufptr, encoding);
826 : 384 : bufptr += MAX_L10N_DATA;
827 : : }
2368 828 : 32 : localized_abbrev_months[12] = NULL;
829 : 32 : localized_full_months[12] = NULL;
830 : :
6674 831 : 32 : CurrentLCTimeValid = true;
832 : : }
833 : :
834 : :
835 : : #if defined(WIN32) && defined(LC_MESSAGES)
836 : : /*
837 : : * Convert a Windows setlocale() argument to a Unix-style one.
838 : : *
839 : : * Regardless of platform, we install message catalogs under a Unix-style
840 : : * LL[_CC][.ENCODING][@VARIANT] naming convention. Only LC_MESSAGES settings
841 : : * following that style will elicit localized interface strings.
842 : : *
843 : : * Before Visual Studio 2012 (msvcr110.dll), Windows setlocale() accepted "C"
844 : : * (but not "c") and strings of the form <Language>[_<Country>][.<CodePage>],
845 : : * case-insensitive. setlocale() returns the fully-qualified form; for
846 : : * example, setlocale("thaI") returns "Thai_Thailand.874". Internally,
847 : : * setlocale() and _create_locale() select a "locale identifier"[1] and store
848 : : * it in an undocumented _locale_t field. From that LCID, we can retrieve the
849 : : * ISO 639 language and the ISO 3166 country. Character encoding does not
850 : : * matter, because the server and client encodings govern that.
851 : : *
852 : : * Windows Vista introduced the "locale name" concept[2], closely following
853 : : * RFC 4646. Locale identifiers are now deprecated. Starting with Visual
854 : : * Studio 2012, setlocale() accepts locale names in addition to the strings it
855 : : * accepted historically. It does not standardize them; setlocale("Th-tH")
856 : : * returns "Th-tH". setlocale(category, "") still returns a traditional
857 : : * string. Furthermore, msvcr110.dll changed the undocumented _locale_t
858 : : * content to carry locale names instead of locale identifiers.
859 : : *
860 : : * Visual Studio 2015 should still be able to do the same as Visual Studio
861 : : * 2012, but the declaration of locale_name is missing in _locale_t, causing
862 : : * this code compilation to fail, hence this falls back instead on to
863 : : * enumerating all system locales by using EnumSystemLocalesEx to find the
864 : : * required locale name. If the input argument is in Unix-style then we can
865 : : * get ISO Locale name directly by using GetLocaleInfoEx() with LCType as
866 : : * LOCALE_SNAME.
867 : : *
868 : : * This function returns a pointer to a static buffer bearing the converted
869 : : * name or NULL if conversion fails.
870 : : *
871 : : * [1] https://docs.microsoft.com/en-us/windows/win32/intl/locale-identifiers
872 : : * [2] https://docs.microsoft.com/en-us/windows/win32/intl/locale-names
873 : : */
874 : :
875 : : /*
876 : : * Callback function for EnumSystemLocalesEx() in get_iso_localename().
877 : : *
878 : : * This function enumerates all system locales, searching for one that matches
879 : : * an input with the format: <Language>[_<Country>], e.g.
880 : : * English[_United States]
881 : : *
882 : : * The input is a three wchar_t array as an LPARAM. The first element is the
883 : : * locale_name we want to match, the second element is an allocated buffer
884 : : * where the Unix-style locale is copied if a match is found, and the third
885 : : * element is the search status, 1 if a match was found, 0 otherwise.
886 : : */
887 : : static BOOL CALLBACK
888 : : search_locale_enum(LPWSTR pStr, DWORD dwFlags, LPARAM lparam)
889 : : {
890 : : wchar_t test_locale[LOCALE_NAME_MAX_LENGTH];
891 : : wchar_t **argv;
892 : :
893 : : (void) (dwFlags);
894 : :
895 : : argv = (wchar_t **) lparam;
896 : : *argv[2] = (wchar_t) 0;
897 : :
898 : : memset(test_locale, 0, sizeof(test_locale));
899 : :
900 : : /* Get the name of the <Language> in English */
901 : : if (GetLocaleInfoEx(pStr, LOCALE_SENGLISHLANGUAGENAME,
902 : : test_locale, LOCALE_NAME_MAX_LENGTH))
903 : : {
904 : : /*
905 : : * If the enumerated locale does not have a hyphen ("en") OR the
906 : : * locale_name input does not have an underscore ("English"), we only
907 : : * need to compare the <Language> tags.
908 : : */
909 : : if (wcsrchr(pStr, '-') == NULL || wcsrchr(argv[0], '_') == NULL)
910 : : {
911 : : if (_wcsicmp(argv[0], test_locale) == 0)
912 : : {
913 : : wcscpy(argv[1], pStr);
914 : : *argv[2] = (wchar_t) 1;
915 : : return FALSE;
916 : : }
917 : : }
918 : :
919 : : /*
920 : : * We have to compare a full <Language>_<Country> tag, so we append
921 : : * the underscore and name of the country/region in English, e.g.
922 : : * "English_United States".
923 : : */
924 : : else
925 : : {
926 : : size_t len;
927 : :
928 : : wcscat(test_locale, L"_");
929 : : len = wcslen(test_locale);
930 : : if (GetLocaleInfoEx(pStr, LOCALE_SENGLISHCOUNTRYNAME,
931 : : test_locale + len,
932 : : LOCALE_NAME_MAX_LENGTH - len))
933 : : {
934 : : if (_wcsicmp(argv[0], test_locale) == 0)
935 : : {
936 : : wcscpy(argv[1], pStr);
937 : : *argv[2] = (wchar_t) 1;
938 : : return FALSE;
939 : : }
940 : : }
941 : : }
942 : : }
943 : :
944 : : return TRUE;
945 : : }
946 : :
947 : : /*
948 : : * This function converts a Windows locale name to an ISO formatted version
949 : : * for Visual Studio 2015 or greater.
950 : : *
951 : : * Returns NULL, if no valid conversion was found.
952 : : */
953 : : static char *
954 : : get_iso_localename(const char *winlocname)
955 : : {
956 : : wchar_t wc_locale_name[LOCALE_NAME_MAX_LENGTH];
957 : : wchar_t buffer[LOCALE_NAME_MAX_LENGTH];
958 : : static char iso_lc_messages[LOCALE_NAME_MAX_LENGTH];
959 : : const char *period;
960 : : int len;
961 : : int ret_val;
962 : :
963 : : /*
964 : : * Valid locales have the following syntax:
965 : : * <Language>[_<Country>[.<CodePage>]]
966 : : *
967 : : * GetLocaleInfoEx can only take locale name without code-page and for the
968 : : * purpose of this API the code-page doesn't matter.
969 : : */
970 : : period = strchr(winlocname, '.');
971 : : if (period != NULL)
972 : : len = period - winlocname;
973 : : else
974 : : len = pg_mbstrlen(winlocname);
975 : :
976 : : memset(wc_locale_name, 0, sizeof(wc_locale_name));
977 : : memset(buffer, 0, sizeof(buffer));
978 : : MultiByteToWideChar(CP_ACP, 0, winlocname, len, wc_locale_name,
979 : : LOCALE_NAME_MAX_LENGTH);
980 : :
981 : : /*
982 : : * If the lc_messages is already a Unix-style string, we have a direct
983 : : * match with LOCALE_SNAME, e.g. en-US, en_US.
984 : : */
985 : : ret_val = GetLocaleInfoEx(wc_locale_name, LOCALE_SNAME, (LPWSTR) &buffer,
986 : : LOCALE_NAME_MAX_LENGTH);
987 : : if (!ret_val)
988 : : {
989 : : /*
990 : : * Search for a locale in the system that matches language and country
991 : : * name.
992 : : */
993 : : wchar_t *argv[3];
994 : :
995 : : argv[0] = wc_locale_name;
996 : : argv[1] = buffer;
997 : : argv[2] = (wchar_t *) &ret_val;
998 : : EnumSystemLocalesEx(search_locale_enum, LOCALE_WINDOWS, (LPARAM) argv,
999 : : NULL);
1000 : : }
1001 : :
1002 : : if (ret_val)
1003 : : {
1004 : : size_t rc;
1005 : : char *hyphen;
1006 : :
1007 : : /* Locale names use only ASCII, any conversion locale suffices. */
1008 : : rc = wchar2char(iso_lc_messages, buffer, sizeof(iso_lc_messages), NULL);
1009 : : if (rc == -1 || rc == sizeof(iso_lc_messages))
1010 : : return NULL;
1011 : :
1012 : : /*
1013 : : * Since the message catalogs sit on a case-insensitive filesystem, we
1014 : : * need not standardize letter case here. So long as we do not ship
1015 : : * message catalogs for which it would matter, we also need not
1016 : : * translate the script/variant portion, e.g. uz-Cyrl-UZ to
1017 : : * uz_UZ@cyrillic. Simply replace the hyphen with an underscore.
1018 : : */
1019 : : hyphen = strchr(iso_lc_messages, '-');
1020 : : if (hyphen)
1021 : : *hyphen = '_';
1022 : : return iso_lc_messages;
1023 : : }
1024 : :
1025 : : return NULL;
1026 : : }
1027 : :
1028 : : static char *
1029 : : IsoLocaleName(const char *winlocname)
1030 : : {
1031 : : static char iso_lc_messages[LOCALE_NAME_MAX_LENGTH];
1032 : :
1033 : : if (pg_strcasecmp("c", winlocname) == 0 ||
1034 : : pg_strcasecmp("posix", winlocname) == 0)
1035 : : {
1036 : : strcpy(iso_lc_messages, "C");
1037 : : return iso_lc_messages;
1038 : : }
1039 : : else
1040 : : return get_iso_localename(winlocname);
1041 : : }
1042 : :
1043 : : #endif /* WIN32 && LC_MESSAGES */
1044 : :
1045 : : /*
1046 : : * Create a new pg_locale_t struct for the given collation oid.
1047 : : */
1048 : : static pg_locale_t
671 jdavis@postgresql.or 1049 : 232 : create_pg_locale(Oid collid, MemoryContext context)
1050 : : {
1051 : : HeapTuple tp;
1052 : : Form_pg_collation collform;
1053 : : pg_locale_t result;
1054 : : Datum datum;
1055 : : bool isnull;
1056 : :
1057 : 232 : tp = SearchSysCache1(COLLOID, ObjectIdGetDatum(collid));
1058 [ - + ]: 232 : if (!HeapTupleIsValid(tp))
671 jdavis@postgresql.or 1059 [ # # ]:UBC 0 : elog(ERROR, "cache lookup failed for collation %u", collid);
671 jdavis@postgresql.or 1060 :CBC 232 : collform = (Form_pg_collation) GETSTRUCT(tp);
1061 : :
1062 [ + + ]: 232 : if (collform->collprovider == COLLPROVIDER_BUILTIN)
633 1063 : 47 : result = create_pg_locale_builtin(collid, context);
671 1064 [ + + ]: 185 : else if (collform->collprovider == COLLPROVIDER_ICU)
633 1065 : 128 : result = create_pg_locale_icu(collid, context);
671 1066 [ + - ]: 57 : else if (collform->collprovider == COLLPROVIDER_LIBC)
633 1067 : 57 : result = create_pg_locale_libc(collid, context);
1068 : : else
1069 : : /* shouldn't happen */
671 jdavis@postgresql.or 1070 [ # # ]:UBC 0 : PGLOCALE_SUPPORT_ERROR(collform->collprovider);
1071 : :
633 jdavis@postgresql.or 1072 :CBC 228 : result->is_default = false;
1073 : :
596 1074 [ + + - + : 228 : Assert((result->collate_is_c && result->collate == NULL) ||
+ - - + ]
1075 : : (!result->collate_is_c && result->collate != NULL));
1076 : :
422 1077 [ + + - + : 228 : Assert((result->ctype_is_c && result->ctype == NULL) ||
+ - - + ]
1078 : : (!result->ctype_is_c && result->ctype != NULL));
1079 : :
671 1080 : 228 : datum = SysCacheGetAttr(COLLOID, tp, Anum_pg_collation_collversion,
1081 : : &isnull);
1082 [ + + ]: 228 : if (!isnull)
1083 : : {
1084 : : char *actual_versionstr;
1085 : : char *collversionstr;
1086 : :
1087 : 171 : collversionstr = TextDatumGetCString(datum);
1088 : :
1089 [ - + ]: 171 : if (collform->collprovider == COLLPROVIDER_LIBC)
671 jdavis@postgresql.or 1090 :UBC 0 : datum = SysCacheGetAttrNotNull(COLLOID, tp, Anum_pg_collation_collcollate);
1091 : : else
671 jdavis@postgresql.or 1092 :CBC 171 : datum = SysCacheGetAttrNotNull(COLLOID, tp, Anum_pg_collation_colllocale);
1093 : :
1094 : 171 : actual_versionstr = get_collation_actual_version(collform->collprovider,
1095 : 171 : TextDatumGetCString(datum));
1096 [ - + ]: 171 : if (!actual_versionstr)
1097 : : {
1098 : : /*
1099 : : * This could happen when specifying a version in CREATE COLLATION
1100 : : * but the provider does not support versioning, or manually
1101 : : * creating a mess in the catalogs.
1102 : : */
671 jdavis@postgresql.or 1103 [ # # ]:UBC 0 : ereport(ERROR,
1104 : : (errmsg("collation \"%s\" has no actual version, but a version was recorded",
1105 : : NameStr(collform->collname))));
1106 : : }
1107 : :
671 jdavis@postgresql.or 1108 [ - + ]:CBC 171 : if (strcmp(actual_versionstr, collversionstr) != 0)
671 jdavis@postgresql.or 1109 [ # # ]:UBC 0 : ereport(WARNING,
1110 : : (errmsg("collation \"%s\" has version mismatch",
1111 : : NameStr(collform->collname)),
1112 : : errdetail("The collation in the database was created using version %s, "
1113 : : "but the operating system provides version %s.",
1114 : : collversionstr, actual_versionstr),
1115 : : errhint("Rebuild all objects affected by this collation and run "
1116 : : "ALTER COLLATION %s REFRESH VERSION, "
1117 : : "or build PostgreSQL with the right library version.",
1118 : : quote_qualified_identifier(get_namespace_name(collform->collnamespace),
1119 : : NameStr(collform->collname)))));
1120 : : }
1121 : :
671 jdavis@postgresql.or 1122 :CBC 228 : ReleaseSysCache(tp);
1123 : :
1124 : 228 : return result;
1125 : : }
1126 : :
1127 : : /*
1128 : : * Initialize default_locale with database locale settings.
1129 : : */
1130 : : void
760 1131 : 17334 : init_database_collation(void)
1132 : : {
1133 : : HeapTuple tup;
1134 : : Form_pg_database dbform;
1135 : : pg_locale_t result;
1136 : :
633 1137 [ - + ]: 17334 : Assert(default_locale == NULL);
1138 : :
1139 : : /* Fetch our pg_database row normally, via syscache */
760 1140 : 17334 : tup = SearchSysCache1(DATABASEOID, ObjectIdGetDatum(MyDatabaseId));
1141 [ - + ]: 17334 : if (!HeapTupleIsValid(tup))
760 jdavis@postgresql.or 1142 [ # # ]:UBC 0 : elog(ERROR, "cache lookup failed for database %u", MyDatabaseId);
760 jdavis@postgresql.or 1143 :CBC 17334 : dbform = (Form_pg_database) GETSTRUCT(tup);
1144 : :
1145 [ + + ]: 17334 : if (dbform->datlocprovider == COLLPROVIDER_BUILTIN)
633 1146 : 947 : result = create_pg_locale_builtin(DEFAULT_COLLATION_OID,
1147 : : TopMemoryContext);
760 1148 [ + + ]: 16387 : else if (dbform->datlocprovider == COLLPROVIDER_ICU)
633 1149 : 13 : result = create_pg_locale_icu(DEFAULT_COLLATION_OID,
1150 : : TopMemoryContext);
702 1151 [ + - ]: 16374 : else if (dbform->datlocprovider == COLLPROVIDER_LIBC)
633 1152 : 16374 : result = create_pg_locale_libc(DEFAULT_COLLATION_OID,
1153 : : TopMemoryContext);
1154 : : else
1155 : : /* shouldn't happen */
702 jdavis@postgresql.or 1156 [ # # ]:UBC 0 : PGLOCALE_SUPPORT_ERROR(dbform->datlocprovider);
1157 : :
633 jdavis@postgresql.or 1158 :CBC 17332 : result->is_default = true;
1159 : :
316 1160 [ + + - + : 17332 : Assert((result->collate_is_c && result->collate == NULL) ||
+ - - + ]
1161 : : (!result->collate_is_c && result->collate != NULL));
1162 : :
1163 [ + + - + : 17332 : Assert((result->ctype_is_c && result->ctype == NULL) ||
+ - - + ]
1164 : : (!result->ctype_is_c && result->ctype != NULL));
1165 : :
760 1166 : 17332 : ReleaseSysCache(tup);
1167 : :
633 1168 : 17332 : default_locale = result;
760 1169 : 17332 : }
1170 : :
1171 : : /*
1172 : : * Get database default locale.
1173 : : */
1174 : : pg_locale_t
313 1175 : 1848201 : pg_database_locale(void)
1176 : : {
1177 : 1848201 : return pg_newlocale_from_collation(DEFAULT_COLLATION_OID);
1178 : : }
1179 : :
1180 : : /*
1181 : : * Create a pg_locale_t from a collation OID. Results are cached for the
1182 : : * lifetime of the backend. Thus, do not free the result with freelocale().
1183 : : *
1184 : : * For simplicity, we always generate COLLATE + CTYPE even though we
1185 : : * might only need one of them. Since this is called only once per session,
1186 : : * it shouldn't cost much.
1187 : : */
1188 : : pg_locale_t
5679 peter_e@gmx.net 1189 : 17644507 : pg_newlocale_from_collation(Oid collid)
1190 : : {
1191 : : collation_cache_entry *cache_entry;
1192 : : bool found;
1193 : :
1194 [ + + ]: 17644507 : if (collid == DEFAULT_COLLATION_OID)
1195 : : {
1196 : : /* should not happen: init_database_collation() not yet run */
80 jdavis@postgresql.or 1197 [ - + ]: 14894586 : if (default_locale == NULL)
80 jdavis@postgresql.or 1198 [ # # ]:UBC 0 : elog(ERROR, "default locale not initialized");
1199 : :
633 jdavis@postgresql.or 1200 :CBC 14894586 : return default_locale;
1201 : : }
1202 : :
1203 : : /*
1204 : : * Some callers expect C_COLLATION_OID to succeed even without catalog
1205 : : * access.
1206 : : */
296 1207 [ + + ]: 2749921 : if (collid == C_COLLATION_OID)
1208 : 2715229 : return &c_locale;
1209 : :
722 1210 [ - + ]: 34692 : if (!OidIsValid(collid))
722 jdavis@postgresql.or 1211 [ # # ]:UBC 0 : elog(ERROR, "cache lookup failed for collation %u", collid);
1212 : :
497 noah@leadboat.com 1213 :CBC 34692 : AssertCouldGetRelation();
1214 : :
723 jdavis@postgresql.or 1215 [ + + ]: 34692 : if (last_collation_cache_oid == collid)
1216 : 33740 : return last_collation_cache_locale;
1217 : :
671 1218 [ + + ]: 952 : if (CollationCache == NULL)
1219 : : {
1220 : 58 : CollationCacheContext = AllocSetContextCreate(TopMemoryContext,
1221 : : "collation cache",
1222 : : ALLOCSET_DEFAULT_SIZES);
1223 : 58 : CollationCache = collation_cache_create(CollationCacheContext,
1224 : : 16, NULL);
1225 : : }
1226 : :
1227 : 952 : cache_entry = collation_cache_insert(CollationCache, collid, &found);
1228 [ + + ]: 952 : if (!found)
1229 : : {
1230 : : /*
1231 : : * Make sure cache entry is marked invalid, in case we fail before
1232 : : * setting things.
1233 : : */
268 peter@eisentraut.org 1234 : 232 : cache_entry->locale = NULL;
1235 : : }
1236 : :
1237 [ + + ]: 952 : if (cache_entry->locale == NULL)
1238 : : {
671 jdavis@postgresql.or 1239 : 232 : cache_entry->locale = create_pg_locale(collid, CollationCacheContext);
1240 : : }
1241 : :
723 1242 : 948 : last_collation_cache_oid = collid;
1243 : 948 : last_collation_cache_locale = cache_entry->locale;
1244 : :
5639 tgl@sss.pgh.pa.us 1245 : 948 : return cache_entry->locale;
1246 : : }
1247 : :
1248 : : /*
1249 : : * Get provider-specific collation version string for the given collation from
1250 : : * the operating system/library.
1251 : : */
1252 : : char *
2008 tmunro@postgresql.or 1253 : 63833 : get_collation_actual_version(char collprovider, const char *collcollate)
1254 : : {
2507 1255 : 63833 : char *collversion = NULL;
1256 : :
897 jdavis@postgresql.or 1257 [ + + ]: 63833 : if (collprovider == COLLPROVIDER_BUILTIN)
596 1258 : 1032 : collversion = get_collation_actual_version_builtin(collcollate);
1259 : : #ifdef USE_ICU
1260 [ + + ]: 62801 : else if (collprovider == COLLPROVIDER_ICU)
1261 : 46409 : collversion = get_collation_actual_version_icu(collcollate);
1262 : : #endif
1263 [ + - ]: 16392 : else if (collprovider == COLLPROVIDER_LIBC)
1264 : 16392 : collversion = get_collation_actual_version_libc(collcollate);
1265 : :
3444 peter_e@gmx.net 1266 : 63833 : return collversion;
1267 : : }
1268 : :
1269 : : /* lowercasing/casefolding in C locale */
1270 : : static size_t
104 jdavis@postgresql.or 1271 : 4395212 : strlower_c(char *dst, size_t dstsize, const char *src, size_t srclen)
1272 : : {
1273 : : size_t i;
1274 : :
274 1275 [ + + + + ]: 35468645 : for (i = 0; i < srclen && i < dstsize; i++)
1276 : 31073433 : dst[i] = pg_ascii_tolower(src[i]);
1277 [ + + ]: 4395212 : if (i < dstsize)
1278 : 4395204 : dst[i] = '\0';
1279 : 4395212 : return srclen;
1280 : : }
1281 : :
1282 : : /* titlecasing in C locale */
1283 : : static size_t
104 1284 : 8 : strtitle_c(char *dst, size_t dstsize, const char *src, size_t srclen)
1285 : : {
274 1286 : 8 : bool wasalnum = false;
1287 : : size_t i;
1288 : :
1289 [ + + + - ]: 96 : for (i = 0; i < srclen && i < dstsize; i++)
1290 : : {
1291 : 88 : char c = src[i];
1292 : :
1293 [ + + ]: 88 : if (wasalnum)
1294 : 72 : dst[i] = pg_ascii_tolower(c);
1295 : : else
1296 : 16 : dst[i] = pg_ascii_toupper(c);
1297 : :
1298 [ + - + + ]: 176 : wasalnum = ((c >= '0' && c <= '9') ||
1299 [ + + + - : 256 : (c >= 'A' && c <= 'Z') ||
+ + ]
1300 [ + - ]: 80 : (c >= 'a' && c <= 'z'));
1301 : : }
1302 [ + - ]: 8 : if (i < dstsize)
1303 : 8 : dst[i] = '\0';
1304 : 8 : return srclen;
1305 : : }
1306 : :
1307 : : /* uppercasing in C locale */
1308 : : static size_t
104 1309 : 16 : strupper_c(char *dst, size_t dstsize, const char *src, size_t srclen)
1310 : : {
1311 : : size_t i;
1312 : :
274 1313 [ + + + + ]: 40 : for (i = 0; i < srclen && i < dstsize; i++)
1314 : 24 : dst[i] = pg_ascii_toupper(src[i]);
1315 [ + + ]: 16 : if (i < dstsize)
1316 : 8 : dst[i] = '\0';
1317 : 16 : return srclen;
1318 : : }
1319 : :
1320 : : /*
1321 : : * pg_strlower()
1322 : : *
1323 : : * Convert src to lowercase, and return the result length (not including
1324 : : * terminating NUL).
1325 : : *
1326 : : * Lowercasing is intended for human-readable display. If the goal is to
1327 : : * convert to a canonical caseless form, see pg_strfold().
1328 : : *
1329 : : * src must be in the database encoding with no embedded NULs. If dstsize is
1330 : : * zero, dst may be NULL, which is useful for calculating the required buffer
1331 : : * size before allocating.
1332 : : *
1333 : : * If the result length is less than dstsize, the NUL-terminated result is
1334 : : * stored in dst. Otherwise, the contents of dst are undefined, and the
1335 : : * caller should use the return value to resize the buffer and retry.
1336 : : *
1337 : : * See pg_locale.h for limits on string expansion.
1338 : : */
1339 : : size_t
104 1340 : 227739 : pg_strlower(char *dst, size_t dstsize, const char *src, size_t srclen,
1341 : : pg_locale_t locale)
1342 : : {
1343 : : size_t result;
1344 : :
274 1345 [ + + ]: 227739 : if (locale->ctype == NULL)
3 jdavis@postgresql.or 1346 :GNC 16 : result = strlower_c(dst, dstsize, src, srclen);
1347 : : else
1348 : 227723 : result = locale->ctype->strlower(dst, dstsize, src, srclen, locale);
1349 : :
1350 [ - + ]: 227739 : Assert(result <= (uint64) srclen * PG_MAX_CASEMAP_EXPANSION);
1351 : 227739 : return result;
1352 : : }
1353 : :
1354 : : /*
1355 : : * pg_strtitle()
1356 : : *
1357 : : * Convert src to titlecase, and return the result length (not including
1358 : : * terminating NUL).
1359 : : *
1360 : : * Titlecasing is intended for human-readable display. A titlecase string has
1361 : : * the initial letter of each word uppercased (or changed to a special
1362 : : * titlecase form, if available), and all other characters lowercased. Used
1363 : : * to implement the SQL INITCAP() function.
1364 : : *
1365 : : * src must be in the database encoding with no embedded NULs. If dstsize is
1366 : : * zero, dst may be NULL, which is useful for calculating the required buffer
1367 : : * size before allocating.
1368 : : *
1369 : : * If the result length is less than dstsize, the NUL-terminated result is
1370 : : * stored in dst. Otherwise, the contents of dst are undefined, and the
1371 : : * caller should use the return value to resize the buffer and retry.
1372 : : *
1373 : : * See pg_locale.h for limits on string expansion.
1374 : : */
1375 : : size_t
104 jdavis@postgresql.or 1376 :CBC 187 : pg_strtitle(char *dst, size_t dstsize, const char *src, size_t srclen,
1377 : : pg_locale_t locale)
1378 : : {
1379 : : size_t result;
1380 : :
274 1381 [ + + ]: 187 : if (locale->ctype == NULL)
3 jdavis@postgresql.or 1382 :GNC 8 : result = strtitle_c(dst, dstsize, src, srclen);
1383 : : else
1384 : 179 : result = locale->ctype->strtitle(dst, dstsize, src, srclen, locale);
1385 : :
1386 [ - + ]: 187 : Assert(result <= (uint64) srclen * PG_MAX_CASEMAP_EXPANSION);
1387 : 187 : return result;
1388 : : }
1389 : :
1390 : : /*
1391 : : * pg_strupper()
1392 : : *
1393 : : * Convert src to uppercase, and return the result length (not including
1394 : : * terminating NUL).
1395 : : *
1396 : : * Uppercasing is intended for human-readable display. If the goal is to
1397 : : * convert to a canonical caseless form, see pg_strfold().
1398 : : *
1399 : : * src must be in the database encoding with no embedded NULs. If dstsize is
1400 : : * zero, dst may be NULL, which is useful for calculating the required buffer
1401 : : * size before allocating.
1402 : : *
1403 : : * If the result length is less than dstsize, the NUL-terminated result is
1404 : : * stored in dst. Otherwise, the contents of dst are undefined, and the
1405 : : * caller should use the return value to resize the buffer and retry.
1406 : : *
1407 : : * See pg_locale.h for limits on string expansion.
1408 : : */
1409 : : size_t
104 jdavis@postgresql.or 1410 :CBC 681133 : pg_strupper(char *dst, size_t dstsize, const char *src, size_t srclen,
1411 : : pg_locale_t locale)
1412 : : {
1413 : : size_t result;
1414 : :
274 1415 [ + + ]: 681133 : if (locale->ctype == NULL)
3 jdavis@postgresql.or 1416 :GNC 16 : result = strupper_c(dst, dstsize, src, srclen);
1417 : : else
1418 : 681117 : result = locale->ctype->strupper(dst, dstsize, src, srclen, locale);
1419 : :
1420 [ - + ]: 681133 : Assert(result <= (uint64) srclen * PG_MAX_CASEMAP_EXPANSION);
1421 : 681133 : return result;
1422 : : }
1423 : :
1424 : : /*
1425 : : * pg_strfold()
1426 : : *
1427 : : * Casefold src, and return the result length (not including terminating
1428 : : * NUL).
1429 : : *
1430 : : * Casefolding produces a canonical string such that, iff the casefolded
1431 : : * strings are equal, the original strings are a case-insensitive match (the
1432 : : * strength of this guarantee depends on normalization, provider and locale).
1433 : : * In practice the result is similar to lowercasing, but the purpose is
1434 : : * different: lowercasing is for human-readable display; whereas casefolding
1435 : : * is meant to canonicalize complex mappings reliably without regard for
1436 : : * display. Unicode guarantees that casefolding is stable across versions if
1437 : : * the original string consists only of assigned code points.
1438 : : *
1439 : : * src must be in the database encoding with no embedded NULs. If dstsize is
1440 : : * zero, dst may be NULL, which is useful for calculating the required buffer
1441 : : * size before allocating.
1442 : : *
1443 : : * If the result length is less than dstsize, the NUL-terminated result is
1444 : : * stored in dst. Otherwise, the contents of dst are undefined, and the
1445 : : * caller should use the return value to resize the buffer and retry.
1446 : : *
1447 : : * See pg_locale.h for limits on string expansion.
1448 : : */
1449 : : size_t
104 jdavis@postgresql.or 1450 :CBC 298174 : pg_strfold(char *dst, size_t dstsize, const char *src, size_t srclen,
1451 : : pg_locale_t locale)
1452 : : {
1453 : : size_t result;
1454 : :
1455 : : /* in the C locale, casefolding is the same as lowercasing */
274 1456 [ + + ]: 298174 : if (locale->ctype == NULL)
3 jdavis@postgresql.or 1457 :GNC 8 : result = strlower_c(dst, dstsize, src, srclen);
1458 : : else
1459 : 298166 : result = locale->ctype->strfold(dst, dstsize, src, srclen, locale);
1460 : :
1461 [ - + ]: 298174 : Assert(result <= (uint64) srclen * PG_MAX_CASEMAP_EXPANSION);
1462 : 298174 : return result;
1463 : : }
1464 : :
1465 : : /*
1466 : : * pg_downcase_ident()
1467 : : *
1468 : : * Lowercase an identifier using historical identifier-folding semantics, and
1469 : : * return the result length (not including terminating NUL). If the result
1470 : : * length is less than dstsize, the NUL-terminated result is stored in dst;
1471 : : * otherwise the contents of dst are undefined.
1472 : : *
1473 : : * XXX: callers currently depend on the result length being equal to srclen,
1474 : : * but that may change in the future if we change to proper case folding.
1475 : : */
1476 : : size_t
104 jdavis@postgresql.or 1477 :CBC 4408467 : pg_downcase_ident(char *dst, size_t dstsize, const char *src, size_t srclen)
1478 : : {
254 1479 : 4408467 : pg_locale_t locale = default_locale;
1480 : :
1481 [ + + + + ]: 4408467 : if (locale == NULL || locale->ctype == NULL ||
1482 [ + + ]: 4184796 : locale->ctype->downcase_ident == NULL)
1483 : 4395188 : return strlower_c(dst, dstsize, src, srclen);
1484 : : else
1485 : 13279 : return locale->ctype->downcase_ident(dst, dstsize, src, srclen,
1486 : : locale);
1487 : : }
1488 : :
1489 : : /*
1490 : : * pg_strcoll
1491 : : *
1492 : : * Like pg_strncoll for NUL-terminated input strings.
1493 : : */
1494 : : int
1281 1495 : 14219891 : pg_strcoll(const char *arg1, const char *arg2, pg_locale_t locale)
1496 : : {
7 1497 [ + + ]: 14219891 : if (locale->collate == NULL)
1498 : 60 : return strcmp(arg1, arg2);
1499 : : else
1500 : 14219831 : return locale->collate->strcoll(arg1, arg2, locale);
1501 : : }
1502 : :
1503 : : /*
1504 : : * pg_strncoll
1505 : : *
1506 : : * Compare strings according to the given locale.
1507 : : *
1508 : : * Strings must be encoded in the database encoding with no embedded NULs.
1509 : : *
1510 : : * The caller is responsible for breaking ties if the collation is
1511 : : * deterministic; this maintains consistency with pg_strnxfrm(), which cannot
1512 : : * easily account for deterministic collations.
1513 : : */
1514 : : int
104 1515 : 2852373 : pg_strncoll(const char *arg1, size_t len1, const char *arg2, size_t len2,
1516 : : pg_locale_t locale)
1517 : : {
7 1518 [ + + ]: 2852373 : if (locale->collate == NULL)
1519 : : {
1520 : 48 : int result = memcmp(arg1, arg2, Min(len1, len2));
1521 : :
1522 [ + + + + ]: 48 : if ((result == 0) && (len1 != len2))
1523 [ + + ]: 24 : result = (len1 < len2) ? -1 : 1;
1524 : 48 : return result;
1525 : : }
1526 : : else
1527 : 2852325 : return locale->collate->strncoll(arg1, len1, arg2, len2, locale);
1528 : : }
1529 : :
1530 : : /*
1531 : : * Return true if the locale supports pg_strxfrm() and pg_strnxfrm();
1532 : : * otherwise false.
1533 : : */
1534 : : bool
1281 1535 : 27573 : pg_strxfrm_enabled(pg_locale_t locale)
1536 : : {
7 1537 [ + + ]: 27573 : if (locale->collate == NULL)
1538 : 12 : return true;
1539 : :
1540 : : /*
1541 : : * locale->collate->strnxfrm is still a required method, even if it may
1542 : : * have the wrong behavior, because the planner uses it for estimates in
1543 : : * some cases.
1544 : : */
596 1545 : 27561 : return locale->collate->strxfrm_is_safe;
1546 : : }
1547 : :
1548 : : /*
1549 : : * pg_strxfrm
1550 : : *
1551 : : * Like pg_strnxfrm for a NUL-terminated input string.
1552 : : */
1553 : : size_t
1281 1554 : 168 : pg_strxfrm(char *dest, const char *src, size_t destsize, pg_locale_t locale)
1555 : : {
7 1556 [ + + ]: 168 : if (locale->collate == NULL)
1557 : 36 : return pg_strnxfrm(dest, destsize, src, strlen(src), locale);
1558 : : else
1559 : 132 : return locale->collate->strxfrm(dest, destsize, src, locale);
1560 : : }
1561 : :
1562 : : /*
1563 : : * pg_strnxfrm
1564 : : *
1565 : : * Transforms 'src' to a nul-terminated string stored in 'dest' such that
1566 : : * ordinary strcmp() on transformed strings is equivalent to pg_strcoll() on
1567 : : * untransformed strings.
1568 : : *
1569 : : * String must be encoded in the database encoding with no embedded NULs. If
1570 : : * 'destsize' is zero, 'dest' may be NULL.
1571 : : *
1572 : : * Not all providers support pg_strnxfrm() safely. The caller should check
1573 : : * pg_strxfrm_enabled() first, otherwise this function may return wrong
1574 : : * results or an error.
1575 : : *
1576 : : * Returns the number of bytes needed (NB: or more; see comments above
1577 : : * strnxfrm_libc()) to store the transformed string, excluding the terminating
1578 : : * nul byte. If the value returned is 'destsize' or greater, the resulting
1579 : : * contents of 'dest' are undefined, and the caller should use the return
1580 : : * value to resize the buffer.
1581 : : */
1582 : : size_t
104 1583 : 8420 : pg_strnxfrm(char *dest, size_t destsize, const char *src, size_t srclen,
1584 : : pg_locale_t locale)
1585 : : {
7 1586 [ + + ]: 8420 : if (locale->collate == NULL)
1587 : : {
1588 [ + + ]: 84 : if (destsize > srclen)
1589 : : {
1590 : 48 : memcpy(dest, src, srclen);
1591 : 48 : dest[srclen] = '\0';
1592 : : }
1593 : :
1594 : 84 : return srclen;
1595 : : }
596 1596 : 8336 : return locale->collate->strnxfrm(dest, destsize, src, srclen, locale);
1597 : : }
1598 : :
1599 : : /*
1600 : : * Return true if the locale supports pg_strxfrm_prefix() and
1601 : : * pg_strnxfrm_prefix(); otherwise false.
1602 : : */
1603 : : bool
1281 1604 : 1318 : pg_strxfrm_prefix_enabled(pg_locale_t locale)
1605 : : {
7 1606 [ + + ]: 1318 : if (locale->collate == NULL)
1607 : 12 : return true;
1608 : : else
1609 : 1306 : return (locale->collate->strnxfrm_prefix != NULL);
1610 : : }
1611 : :
1612 : : /*
1613 : : * pg_strxfrm_prefix
1614 : : *
1615 : : * Like pg_strnxfrm_prefix for a NUL-terminated input string.
1616 : : */
1617 : : size_t
1281 1618 : 1314 : pg_strxfrm_prefix(char *dest, const char *src, size_t destsize,
1619 : : pg_locale_t locale)
1620 : : {
7 1621 [ + + ]: 1314 : if (locale->collate == NULL)
1622 : 12 : return pg_strnxfrm_prefix(dest, destsize, src, strlen(src), locale);
1623 : : else
1624 : 1302 : return locale->collate->strxfrm_prefix(dest, destsize, src, locale);
1625 : : }
1626 : :
1627 : : /*
1628 : : * pg_strnxfrm_prefix
1629 : : *
1630 : : * Transforms 'src' to a byte sequence stored in 'dest' such that ordinary
1631 : : * memcmp() on the byte sequence is equivalent to pg_strncoll() on
1632 : : * untransformed strings. The result is not nul-terminated.
1633 : : *
1634 : : * String must be encoded in the database encoding with no embedded NULs. If
1635 : : * destsize is zero, dest may be NULL.
1636 : : *
1637 : : * Not all providers support pg_strnxfrm_prefix() safely. The caller must
1638 : : * check pg_strxfrm_prefix_enabled() first.
1639 : : *
1640 : : * If destsize is not large enough to hold the resulting byte sequence, stores
1641 : : * only the first destsize bytes in 'dest'. Returns the number of bytes
1642 : : * actually copied to 'dest'.
1643 : : */
1644 : : size_t
1281 1645 : 52 : pg_strnxfrm_prefix(char *dest, size_t destsize, const char *src,
1646 : : size_t srclen, pg_locale_t locale)
1647 : : {
7 1648 [ + + ]: 52 : if (locale->collate == NULL)
1649 : : {
1650 : 48 : size_t len = Min(srclen, destsize);
1651 : :
1652 [ + + ]: 48 : if (destsize > 0)
1653 : 36 : memcpy(dest, src, len);
1654 : 48 : return len;
1655 : : }
1656 : : else
1657 : 4 : return locale->collate->strnxfrm_prefix(dest, destsize, src, srclen, locale);
1658 : : }
1659 : :
1660 : : /*
1661 : : * pg_iswdigit(), pg_iswalpha(), etc.
1662 : : *
1663 : : * Character semantics for pattern-matching. Uses pg_wchar, which is an
1664 : : * encoding-dependent value, which may or may not be equivalent to a code
1665 : : * point.
1666 : : *
1667 : : * For case-mapping an entire string, use pg_strlower(), etc., instead.
1668 : : * String based functions can handle one-to-many and context-sensitive
1669 : : * mappings.
1670 : : */
1671 : :
1672 : : bool
316 1673 : 30258 : pg_iswdigit(pg_wchar wc, pg_locale_t locale)
1674 : : {
1675 [ - + ]: 30258 : if (locale->ctype == NULL)
316 jdavis@postgresql.or 1676 [ # # ]:UBC 0 : return (wc <= (pg_wchar) 127 &&
1677 [ # # ]: 0 : (pg_char_properties[wc] & PG_ISDIGIT));
1678 : : else
316 jdavis@postgresql.or 1679 :CBC 30258 : return locale->ctype->wc_isdigit(wc, locale);
1680 : : }
1681 : :
1682 : : bool
1683 : 83406 : pg_iswalpha(pg_wchar wc, pg_locale_t locale)
1684 : : {
1685 [ - + ]: 83406 : if (locale->ctype == NULL)
316 jdavis@postgresql.or 1686 [ # # ]:UBC 0 : return (wc <= (pg_wchar) 127 &&
1687 [ # # ]: 0 : (pg_char_properties[wc] & PG_ISALPHA));
1688 : : else
316 jdavis@postgresql.or 1689 :CBC 83406 : return locale->ctype->wc_isalpha(wc, locale);
1690 : : }
1691 : :
1692 : : bool
1693 : 1615885 : pg_iswalnum(pg_wchar wc, pg_locale_t locale)
1694 : : {
1695 [ - + ]: 1615885 : if (locale->ctype == NULL)
316 jdavis@postgresql.or 1696 [ # # ]:UBC 0 : return (wc <= (pg_wchar) 127 &&
1697 [ # # ]: 0 : (pg_char_properties[wc] & PG_ISALNUM));
1698 : : else
316 jdavis@postgresql.or 1699 :CBC 1615885 : return locale->ctype->wc_isalnum(wc, locale);
1700 : : }
1701 : :
1702 : : bool
316 jdavis@postgresql.or 1703 :UBC 0 : pg_iswupper(pg_wchar wc, pg_locale_t locale)
1704 : : {
1705 [ # # ]: 0 : if (locale->ctype == NULL)
1706 [ # # ]: 0 : return (wc <= (pg_wchar) 127 &&
1707 [ # # ]: 0 : (pg_char_properties[wc] & PG_ISUPPER));
1708 : : else
1709 : 0 : return locale->ctype->wc_isupper(wc, locale);
1710 : : }
1711 : :
1712 : : bool
1713 : 0 : pg_iswlower(pg_wchar wc, pg_locale_t locale)
1714 : : {
1715 [ # # ]: 0 : if (locale->ctype == NULL)
1716 [ # # ]: 0 : return (wc <= (pg_wchar) 127 &&
1717 [ # # ]: 0 : (pg_char_properties[wc] & PG_ISLOWER));
1718 : : else
1719 : 0 : return locale->ctype->wc_islower(wc, locale);
1720 : : }
1721 : :
1722 : : bool
1723 : 0 : pg_iswgraph(pg_wchar wc, pg_locale_t locale)
1724 : : {
1725 [ # # ]: 0 : if (locale->ctype == NULL)
1726 [ # # ]: 0 : return (wc <= (pg_wchar) 127 &&
1727 [ # # ]: 0 : (pg_char_properties[wc] & PG_ISGRAPH));
1728 : : else
1729 : 0 : return locale->ctype->wc_isgraph(wc, locale);
1730 : : }
1731 : :
1732 : : bool
1733 : 0 : pg_iswprint(pg_wchar wc, pg_locale_t locale)
1734 : : {
1735 [ # # ]: 0 : if (locale->ctype == NULL)
1736 [ # # ]: 0 : return (wc <= (pg_wchar) 127 &&
1737 [ # # ]: 0 : (pg_char_properties[wc] & PG_ISPRINT));
1738 : : else
1739 : 0 : return locale->ctype->wc_isprint(wc, locale);
1740 : : }
1741 : :
1742 : : bool
1743 : 0 : pg_iswpunct(pg_wchar wc, pg_locale_t locale)
1744 : : {
1745 [ # # ]: 0 : if (locale->ctype == NULL)
1746 [ # # ]: 0 : return (wc <= (pg_wchar) 127 &&
1747 [ # # ]: 0 : (pg_char_properties[wc] & PG_ISPUNCT));
1748 : : else
1749 : 0 : return locale->ctype->wc_ispunct(wc, locale);
1750 : : }
1751 : :
1752 : : bool
316 jdavis@postgresql.or 1753 :CBC 502 : pg_iswspace(pg_wchar wc, pg_locale_t locale)
1754 : : {
1755 [ - + ]: 502 : if (locale->ctype == NULL)
316 jdavis@postgresql.or 1756 [ # # ]:UBC 0 : return (wc <= (pg_wchar) 127 &&
1757 [ # # ]: 0 : (pg_char_properties[wc] & PG_ISSPACE));
1758 : : else
316 jdavis@postgresql.or 1759 :CBC 502 : return locale->ctype->wc_isspace(wc, locale);
1760 : : }
1761 : :
1762 : : bool
313 1763 : 12 : pg_iswxdigit(pg_wchar wc, pg_locale_t locale)
1764 : : {
1765 [ - + ]: 12 : if (locale->ctype == NULL)
313 jdavis@postgresql.or 1766 [ # # ]:UBC 0 : return (wc <= (pg_wchar) 127 &&
1767 [ # # # # ]: 0 : ((pg_char_properties[wc] & PG_ISDIGIT) ||
1768 [ # # # # ]: 0 : ((wc >= 'A' && wc <= 'F') ||
1769 [ # # ]: 0 : (wc >= 'a' && wc <= 'f'))));
1770 : : else
313 jdavis@postgresql.or 1771 :CBC 12 : return locale->ctype->wc_isxdigit(wc, locale);
1772 : : }
1773 : :
1774 : : bool
260 1775 : 219 : pg_iswcased(pg_wchar wc, pg_locale_t locale)
1776 : : {
1777 : : /* for the C locale, Cased and Alpha are equivalent */
1778 [ + + ]: 219 : if (locale->ctype == NULL)
1779 [ + - ]: 228 : return (wc <= (pg_wchar) 127 &&
1780 [ + + ]: 114 : (pg_char_properties[wc] & PG_ISALPHA));
1781 : : else
1782 : 105 : return locale->ctype->wc_iscased(wc, locale);
1783 : : }
1784 : :
1785 : : pg_wchar
316 jdavis@postgresql.or 1786 :UBC 0 : pg_towupper(pg_wchar wc, pg_locale_t locale)
1787 : : {
1788 [ # # ]: 0 : if (locale->ctype == NULL)
1789 : : {
1790 [ # # ]: 0 : if (wc <= (pg_wchar) 127)
1791 : 0 : return pg_ascii_toupper((unsigned char) wc);
1792 : 0 : return wc;
1793 : : }
1794 : : else
1795 : 0 : return locale->ctype->wc_toupper(wc, locale);
1796 : : }
1797 : :
1798 : : pg_wchar
1799 : 0 : pg_towlower(pg_wchar wc, pg_locale_t locale)
1800 : : {
1801 [ # # ]: 0 : if (locale->ctype == NULL)
1802 : : {
1803 [ # # ]: 0 : if (wc <= (pg_wchar) 127)
1804 : 0 : return pg_ascii_tolower((unsigned char) wc);
1805 : 0 : return wc;
1806 : : }
1807 : : else
1808 : 0 : return locale->ctype->wc_tolower(wc, locale);
1809 : : }
1810 : :
1811 : : /* version of Unicode used by ICU */
1812 : : const char *
178 jdavis@postgresql.or 1813 :CBC 1 : pg_icu_unicode_version(void)
1814 : : {
1815 : : #ifdef USE_ICU
233 1816 : 1 : return U_UNICODE_VERSION;
1817 : : #else
1818 : : return NULL;
1819 : : #endif
1820 : : }
1821 : :
1822 : : /*
1823 : : * Return required encoding ID for the given locale, or -1 if any encoding is
1824 : : * valid for the locale.
1825 : : */
1826 : : int
892 1827 : 1074 : builtin_locale_encoding(const char *locale)
1828 : : {
891 1829 [ + + ]: 1074 : if (strcmp(locale, "C") == 0)
1830 : 61 : return -1;
587 1831 [ + + ]: 1013 : else if (strcmp(locale, "C.UTF-8") == 0)
891 1832 : 994 : return PG_UTF8;
587 1833 [ + - ]: 19 : else if (strcmp(locale, "PG_UNICODE_FAST") == 0)
1834 : 19 : return PG_UTF8;
1835 : :
1836 : :
891 jdavis@postgresql.or 1837 [ # # ]:UBC 0 : ereport(ERROR,
1838 : : (errcode(ERRCODE_WRONG_OBJECT_TYPE),
1839 : : errmsg("invalid locale name \"%s\" for builtin provider",
1840 : : locale)));
1841 : :
1842 : : return 0; /* keep compiler quiet */
1843 : : }
1844 : :
1845 : :
1846 : : /*
1847 : : * Validate the locale and encoding combination, and return the canonical form
1848 : : * of the locale name.
1849 : : */
1850 : : const char *
897 jdavis@postgresql.or 1851 :CBC 1060 : builtin_validate_locale(int encoding, const char *locale)
1852 : : {
891 1853 : 1060 : const char *canonical_name = NULL;
1854 : : int required_encoding;
1855 : :
1856 [ + + ]: 1060 : if (strcmp(locale, "C") == 0)
1857 : 49 : canonical_name = "C";
1858 [ + + + + ]: 1011 : else if (strcmp(locale, "C.UTF-8") == 0 || strcmp(locale, "C.UTF8") == 0)
1859 : 985 : canonical_name = "C.UTF-8";
587 1860 [ + + ]: 26 : else if (strcmp(locale, "PG_UNICODE_FAST") == 0)
1861 : 14 : canonical_name = "PG_UNICODE_FAST";
1862 : :
891 1863 [ + + ]: 1060 : if (!canonical_name)
897 1864 [ + - ]: 12 : ereport(ERROR,
1865 : : (errcode(ERRCODE_WRONG_OBJECT_TYPE),
1866 : : errmsg("invalid locale name \"%s\" for builtin provider",
1867 : : locale)));
1868 : :
891 1869 : 1048 : required_encoding = builtin_locale_encoding(canonical_name);
1870 [ + + + + ]: 1048 : if (required_encoding >= 0 && encoding != required_encoding)
1871 [ + - ]: 1 : ereport(ERROR,
1872 : : (errcode(ERRCODE_WRONG_OBJECT_TYPE),
1873 : : errmsg("encoding \"%s\" does not match locale \"%s\"",
1874 : : pg_encoding_to_char(encoding), locale)));
1875 : :
1876 : 1047 : return canonical_name;
1877 : : }
1878 : :
1879 : :
1880 : :
1881 : : /*
1882 : : * Return the BCP47 language tag representation of the requested locale.
1883 : : *
1884 : : * This function should be called before passing the string to ucol_open(),
1885 : : * because conversion to a language tag also performs "level 2
1886 : : * canonicalization". In addition to producing a consistent format, level 2
1887 : : * canonicalization is able to more accurately interpret different input
1888 : : * locale string formats, such as POSIX and .NET IDs.
1889 : : */
1890 : : char *
1241 1891 : 46216 : icu_language_tag(const char *loc_str, int elevel)
1892 : : {
1893 : : #ifdef USE_ICU
1894 : : UErrorCode status;
1895 : : char *langtag;
1196 tgl@sss.pgh.pa.us 1896 : 46216 : size_t buflen = 32; /* arbitrary starting buffer size */
1897 : 46216 : const bool strict = true;
1898 : :
1899 : : /*
1900 : : * A BCP47 language tag doesn't have a clearly-defined upper limit (cf.
1901 : : * RFC5646 section 4.4). Additionally, in older ICU versions,
1902 : : * uloc_toLanguageTag() doesn't always return the ultimate length on the
1903 : : * first call, necessitating a loop.
1904 : : */
1241 jdavis@postgresql.or 1905 : 46216 : langtag = palloc(buflen);
1906 : : while (true)
1907 : : {
1908 : 46216 : status = U_ZERO_ERROR;
1198 1909 : 46216 : uloc_toLanguageTag(loc_str, langtag, buflen, strict, &status);
1910 : :
1911 : : /* try again if the buffer is not large enough */
1241 1912 [ + - ]: 46216 : if ((status == U_BUFFER_OVERFLOW_ERROR ||
1198 1913 [ - + - - ]: 46216 : status == U_STRING_NOT_TERMINATED_WARNING) &&
1914 : : buflen < MaxAllocSize)
1915 : : {
1241 jdavis@postgresql.or 1916 :UBC 0 : buflen = Min(buflen * 2, MaxAllocSize);
1917 : 0 : langtag = repalloc(langtag, buflen);
1918 : 0 : continue;
1919 : : }
1920 : :
1241 jdavis@postgresql.or 1921 :CBC 46216 : break;
1922 : : }
1923 : :
1924 [ + + ]: 46216 : if (U_FAILURE(status))
1925 : : {
1926 : 11 : pfree(langtag);
1927 : :
1928 [ + + ]: 11 : if (elevel > 0)
1929 [ + - ]: 9 : ereport(elevel,
1930 : : (errcode(ERRCODE_INVALID_PARAMETER_VALUE),
1931 : : errmsg("could not convert locale name \"%s\" to language tag: %s",
1932 : : loc_str, u_errorName(status))));
1933 : 7 : return NULL;
1934 : : }
1935 : :
1936 : 46205 : return langtag;
1937 : : #else /* not USE_ICU */
1938 : : ereport(ERROR,
1939 : : (errcode(ERRCODE_FEATURE_NOT_SUPPORTED),
1940 : : errmsg("ICU is not supported in this build")));
1941 : : return NULL; /* keep compiler quiet */
1942 : : #endif /* not USE_ICU */
1943 : : }
1944 : :
1945 : : /*
1946 : : * Perform best-effort check that the locale is a valid one.
1947 : : */
1948 : : void
1248 1949 : 109 : icu_validate_locale(const char *loc_str)
1950 : : {
1951 : : #ifdef USE_ICU
1952 : : UCollator *collator;
1953 : : UErrorCode status;
1954 : : char lang[ULOC_LANG_CAPACITY];
1196 tgl@sss.pgh.pa.us 1955 : 109 : bool found = false;
1956 : 109 : int elevel = icu_validation_level;
1957 : :
1958 : : /* no validation */
1248 jdavis@postgresql.or 1959 [ + + ]: 109 : if (elevel < 0)
1960 : 8 : return;
1961 : :
1962 : : /* downgrade to WARNING during pg_upgrade */
1963 [ + + - + ]: 101 : if (IsBinaryUpgrade && elevel > WARNING)
1248 jdavis@postgresql.or 1964 :UBC 0 : elevel = WARNING;
1965 : :
1966 : : /* validate that we can extract the language */
1248 jdavis@postgresql.or 1967 :CBC 101 : status = U_ZERO_ERROR;
1968 : 101 : uloc_getLanguage(loc_str, lang, ULOC_LANG_CAPACITY, &status);
1198 1969 [ + - - + ]: 101 : if (U_FAILURE(status) || status == U_STRING_NOT_TERMINATED_WARNING)
1970 : : {
1248 jdavis@postgresql.or 1971 [ # # ]:UBC 0 : ereport(elevel,
1972 : : (errcode(ERRCODE_INVALID_PARAMETER_VALUE),
1973 : : errmsg("could not get language from ICU locale \"%s\": %s",
1974 : : loc_str, u_errorName(status)),
1975 : : errhint("To disable ICU locale validation, set the parameter \"%s\" to \"%s\".",
1976 : : "icu_validation_level", "disabled")));
1977 : 0 : return;
1978 : : }
1979 : :
1980 : : /* check for special language name */
1248 jdavis@postgresql.or 1981 [ + + ]:CBC 101 : if (strcmp(lang, "") == 0 ||
1163 1982 [ + - - + ]: 27 : strcmp(lang, "root") == 0 || strcmp(lang, "und") == 0)
1248 1983 : 74 : found = true;
1984 : :
1985 : : /* search for matching language within ICU */
1986 [ + + + + ]: 10384 : for (int32_t i = 0; !found && i < uloc_countAvailable(); i++)
1987 : : {
1196 tgl@sss.pgh.pa.us 1988 : 10283 : const char *otherloc = uloc_getAvailable(i);
1989 : : char otherlang[ULOC_LANG_CAPACITY];
1990 : :
1248 jdavis@postgresql.or 1991 : 10283 : status = U_ZERO_ERROR;
1992 : 10283 : uloc_getLanguage(otherloc, otherlang, ULOC_LANG_CAPACITY, &status);
1198 1993 [ + - - + ]: 10283 : if (U_FAILURE(status) || status == U_STRING_NOT_TERMINATED_WARNING)
1248 jdavis@postgresql.or 1994 :UBC 0 : continue;
1995 : :
1248 jdavis@postgresql.or 1996 [ + + ]:CBC 10283 : if (strcmp(lang, otherlang) == 0)
1997 : 18 : found = true;
1998 : : }
1999 : :
2000 [ + + ]: 101 : if (!found)
2001 [ + - ]: 9 : ereport(elevel,
2002 : : (errcode(ERRCODE_INVALID_PARAMETER_VALUE),
2003 : : errmsg("ICU locale \"%s\" has unknown language \"%s\"",
2004 : : loc_str, lang),
2005 : : errhint("To disable ICU locale validation, set the parameter \"%s\" to \"%s\".",
2006 : : "icu_validation_level", "disabled")));
2007 : :
2008 : : /* check that it can be opened */
2009 : 97 : collator = pg_ucol_open(loc_str);
1621 peter@eisentraut.org 2010 : 92 : ucol_close(collator);
2011 : : #else /* not USE_ICU */
2012 : : /* could get here if a collation was created by a build with ICU */
2013 : : ereport(ERROR,
2014 : : (errcode(ERRCODE_FEATURE_NOT_SUPPORTED),
2015 : : errmsg("ICU is not supported in this build")));
2016 : : #endif /* not USE_ICU */
2017 : : }
|