Bug Summary

File:builds/wireshark/wireshark/wsutil/str_util.c
Warning:line 1472, column 9
Value stored to 'printable_bytes' is never read

Annotated Source Code

Press '?' to see keyboard shortcuts

clang -cc1 -cc1 -triple x86_64-pc-linux-gnu -analyze -disable-free -clear-ast-before-backend -disable-llvm-verifier -discard-value-names -main-file-name str_util.c -analyzer-checker=core -analyzer-checker=apiModeling -analyzer-checker=unix -analyzer-checker=deadcode -analyzer-checker=security.insecureAPI.UncheckedReturn -analyzer-checker=security.insecureAPI.getpw -analyzer-checker=security.insecureAPI.gets -analyzer-checker=security.insecureAPI.mktemp -analyzer-checker=security.insecureAPI.mkstemp -analyzer-checker=security.insecureAPI.vfork -analyzer-checker=nullability.NullPassedToNonnull -analyzer-checker=nullability.NullReturnedFromNonnull -analyzer-output plist -w -setup-static-analyzer -mrelocation-model pic -pic-level 2 -fhalf-no-semantic-interposition -fno-delete-null-pointer-checks -mframe-pointer=all -relaxed-aliasing -fmath-errno -ffp-contract=on -fno-rounding-math -ffloat16-excess-precision=standard -fbfloat16-excess-precision=standard -mconstructor-aliases -funwind-tables=2 -target-cpu x86-64 -tune-cpu generic -debugger-tuning=gdb -fdebug-compilation-dir=/builds/wireshark/wireshark/build -fcoverage-compilation-dir=/builds/wireshark/wireshark/build -resource-dir /usr/lib/llvm-22/lib/clang/22 -isystem /usr/include/glib-2.0 -isystem /usr/lib/x86_64-linux-gnu/glib-2.0/include -D BUILD_WSUTIL -D CARES_NO_DEPRECATED -D G_DISABLE_DEPRECATED -D G_DISABLE_SINGLE_INCLUDES -D WS_BUILD_DLL -D WS_DEBUG -D WS_DEBUG_UTF_8 -D wsutil_EXPORTS -I /builds/wireshark/wireshark/build -I /builds/wireshark/wireshark -I /builds/wireshark/wireshark/include -I /builds/wireshark/wireshark/build/wsutil -D _GLIBCXX_ASSERTIONS -internal-isystem /usr/lib/llvm-22/lib/clang/22/include -internal-isystem /usr/local/include -internal-isystem /usr/lib/gcc/x86_64-linux-gnu/16/../../../../x86_64-linux-gnu/include -internal-externc-isystem /usr/include/x86_64-linux-gnu -internal-externc-isystem /include -internal-externc-isystem /usr/include -fmacro-prefix-map=/builds/wireshark/wireshark/= -fmacro-prefix-map=/builds/wireshark/wireshark/build/= -fmacro-prefix-map=../= -Wno-format-nonliteral -std=gnu17 -ferror-limit 19 -fvisibility=hidden -fwrapv -fwrapv-pointer -fstrict-flex-arrays=3 -stack-protector 2 -fstack-clash-protection -fcf-protection=full -fgnuc-version=4.2.1 -fskip-odr-check-in-gmf -fexceptions -fcolor-diagnostics -analyzer-output=html -faddrsig -fdwarf2-cfi-asm -o /builds/wireshark/wireshark/sbout/2026-10-08-100355-3663-1 -x c /builds/wireshark/wireshark/wsutil/str_util.c
1/* str_util.c
2 * String utility routines
3 *
4 * Wireshark - Network traffic analyzer
5 * By Gerald Combs <gerald@wireshark.org>
6 * Copyright 1998 Gerald Combs
7 *
8 * SPDX-License-Identifier: GPL-2.0-or-later
9 */
10
11#define _GNU_SOURCE
12#include "config.h"
13#include "str_util.h"
14
15#include <string.h>
16#include <locale.h>
17#include <math.h>
18
19#include <ws_codepoints.h>
20
21#include <wsutil/to_str.h>
22#include <wsutil/strtoi.h>
23#include <wsutil/unicode-utils.h>
24
25struct prefix_parameters {
26 const char * const *prefix; /**< array of prefixes to represent unit multiplication factors. */
27 int prefix_count; /**< number of elements in the prefix array. */
28 int power; /**< multiplication factor between prefixes. */
29 int prefix_offset; /**< index of element within the prefix array for "no prefix". */
30};
31
32static const char hex[16] = { '0', '1', '2', '3', '4', '5', '6', '7',
33 '8', '9', 'A', 'B', 'C', 'D', 'E', 'F' };
34
35/* Given a "flags" value passed into a formatting function, determine which
36 * formatting parameters should apply.
37 */
38static const struct prefix_parameters *
39prefix_parameters_for_flags(uint16_t flags) {
40 static const char * const si_prefixes[] = {" a", " f", " p", " n", " μ", " m", " ", " k", " M", " G", " T", " P", " E"};
41 static const struct prefix_parameters si_parameters = {si_prefixes, G_N_ELEMENTS(si_prefixes)(sizeof (si_prefixes) / sizeof ((si_prefixes)[0])), 1000, 6};
42 static const char * const iec_prefixes[] = {" ", " Ki", " Mi", " Gi", " Ti", " Pi", " Ei"};
43 static const struct prefix_parameters iec_parameters = {iec_prefixes, G_N_ELEMENTS(iec_prefixes)(sizeof (iec_prefixes) / sizeof ((iec_prefixes)[0])), 1024, 0};
44
45 return (flags & FORMAT_SIZE_PREFIX_IEC(1 << 1)) != 0 ? &iec_parameters : &si_parameters;
46}
47
48char *
49wmem_strconcat(wmem_allocator_t *allocator, const char *first, ...)
50{
51 size_t len;
52 va_list args;
53 char *s;
54 char *concat;
55 char *ptr;
56
57 if (!first)
58 return NULL((void*)0);
59
60 len = 1 + strlen(first);
61 va_start(args, first)__builtin_va_start(args, first);
62 while ((s = va_arg(args, char*)__builtin_va_arg(args, char*))) {
63 len += strlen(s);
64 }
65 va_end(args)__builtin_va_end(args);
66
67 ptr = concat = (char *)wmem_alloc(allocator, len);
68
69 ptr = g_stpcpy(ptr, first);
70 va_start(args, first)__builtin_va_start(args, first);
71 while ((s = va_arg(args, char*)__builtin_va_arg(args, char*))) {
72 ptr = g_stpcpy(ptr, s);
73 }
74 va_end(args)__builtin_va_end(args);
75
76 return concat;
77}
78
79char *
80wmem_strjoin(wmem_allocator_t *allocator,
81 const char *separator, const char *first, ...)
82{
83 size_t len;
84 va_list args;
85 size_t separator_len;
86 char *s;
87 char *concat;
88 char *ptr;
89
90 if (!first)
91 return NULL((void*)0);
92
93 if (separator == NULL((void*)0)) {
94 separator = "";
95 }
96
97 separator_len = strlen (separator);
98
99 len = 1 + strlen(first); /* + 1 for null byte */
100 va_start(args, first)__builtin_va_start(args, first);
101 while ((s = va_arg(args, char*)__builtin_va_arg(args, char*))) {
102 len += (separator_len + strlen(s));
103 }
104 va_end(args)__builtin_va_end(args);
105
106 ptr = concat = (char *)wmem_alloc(allocator, len);
107 ptr = g_stpcpy(ptr, first);
108 va_start(args, first)__builtin_va_start(args, first);
109 while ((s = va_arg(args, char*)__builtin_va_arg(args, char*))) {
110 ptr = g_stpcpy(ptr, separator);
111 ptr = g_stpcpy(ptr, s);
112 }
113 va_end(args)__builtin_va_end(args);
114
115 return concat;
116
117}
118
119char *
120wmem_strjoinv(wmem_allocator_t *allocator,
121 const char *separator, char **str_array)
122{
123 char *string = NULL((void*)0);
124
125 ws_return_val_if(!str_array, NULL)do { if (1 && (!str_array)) { ws_log_full("InvalidArg"
, LOG_LEVEL_WARNING, "wsutil/str_util.c", 125, __func__, "invalid argument: %s"
, "!str_array"); return (((void*)0)); } } while (0)
;
126
127 if (separator == NULL((void*)0)) {
128 separator = "";
129 }
130
131 if (str_array[0]) {
132 int i;
133 char *ptr;
134 size_t len, separator_len;
135
136 separator_len = strlen(separator);
137
138 /* Get first part of length. Plus one for null byte. */
139 len = 1 + strlen(str_array[0]);
140 /* Get the full length, including the separators. */
141 for (i = 1; str_array[i] != NULL((void*)0); i++) {
142 len += separator_len;
143 len += strlen(str_array[i]);
144 }
145
146 /* Allocate and build the string. */
147 string = (char *)wmem_alloc(allocator, len);
148 ptr = g_stpcpy(string, str_array[0]);
149 for (i = 1; str_array[i] != NULL((void*)0); i++) {
150 ptr = g_stpcpy(ptr, separator);
151 ptr = g_stpcpy(ptr, str_array[i]);
152 }
153 } else {
154 string = wmem_strdup(allocator, "");
155 }
156
157 return string;
158
159}
160
161char **
162wmem_strsplit(wmem_allocator_t *allocator, const char *src,
163 const char *delimiter, int max_tokens)
164{
165 char *splitted;
166 char *s;
167 unsigned tokens;
168 unsigned sep_len;
169 unsigned i;
170 char **vec;
171
172 if (!src || !delimiter || !delimiter[0])
173 return NULL((void*)0);
174
175 /* An empty string results in an empty vector. */
176 if (!src[0]) {
177 vec = wmem_new0(allocator, char *)((char **)wmem_alloc0((allocator), sizeof(char *)));
178 return vec;
179 }
180
181 splitted = wmem_strdup(allocator, src);
182 sep_len = (unsigned)strlen(delimiter);
183
184 if (max_tokens < 1)
185 max_tokens = INT_MAX2147483647;
186
187 /* Calculate the number of fields. */
188 s = splitted;
189 tokens = 1;
190 while (tokens < (unsigned)max_tokens && (s = strstr(s, delimiter)_Generic (0 ? (s) : (void *) 1, const void *: (const char *) (
strstr (s, delimiter)), default: strstr (s, delimiter))
)) {
191 s += sep_len;
192 tokens++;
193 }
194
195 vec = wmem_alloc_array(allocator, char *, tokens + 1)((char **)wmem_alloc((allocator), (((((tokens + 1)) <= 0) ||
((size_t)sizeof(char *) > (9223372036854775807L / (size_t
)((tokens + 1))))) ? 0 : (sizeof(char *) * ((tokens + 1))))))
;
196
197 /* Populate the array of string tokens. */
198 s = splitted;
199 vec[0] = s;
200 tokens = 1;
201 while (tokens < (unsigned)max_tokens && (s = strstr(s, delimiter)_Generic (0 ? (s) : (void *) 1, const void *: (const char *) (
strstr (s, delimiter)), default: strstr (s, delimiter))
)) {
202 for (i = 0; i < sep_len; i++)
203 s[i] = '\0';
204 s += sep_len;
205 vec[tokens] = s;
206 tokens++;
207
208 }
209
210 vec[tokens] = NULL((void*)0);
211
212 return vec;
213}
214
215/*
216 * wmem_ascii_strdown:
217 * based on g_ascii_strdown.
218 */
219char*
220wmem_ascii_strdown(wmem_allocator_t *allocator, const char *str, ssize_t len)
221{
222 char *result, *s;
223 size_t abs_len;
224
225 g_return_val_if_fail (str != NULL, NULL)do { if ((str != ((void*)0))) { } else { g_return_if_fail_warning
(((gchar*) 0), ((const char*) (__func__)), "str != NULL"); return
(((void*)0)); } } while (0)
;
226
227 abs_len = (len < 0) ? strlen(str) : (size_t)len;
228
229 result = wmem_strndup(allocator, str, abs_len);
230 for (s = result; *s; s++)
231 *s = g_ascii_tolower (*s);
232
233 return result;
234}
235
236int
237ws_xton(char ch)
238{
239 switch (ch) {
240 case '0': return 0;
241 case '1': return 1;
242 case '2': return 2;
243 case '3': return 3;
244 case '4': return 4;
245 case '5': return 5;
246 case '6': return 6;
247 case '7': return 7;
248 case '8': return 8;
249 case '9': return 9;
250 case 'a': case 'A': return 10;
251 case 'b': case 'B': return 11;
252 case 'c': case 'C': return 12;
253 case 'd': case 'D': return 13;
254 case 'e': case 'E': return 14;
255 case 'f': case 'F': return 15;
256 default: return -1;
257 }
258}
259
260/* Convert all ASCII letters to lower case, in place. */
261char *
262ascii_strdown_inplace(char *str)
263{
264 char *s;
265
266 for (s = str; *s; s++)
267 /* What 'g_ascii_tolower (char c)' does, this should be slightly more efficient */
268 *s = g_ascii_isupper (*s)((g_ascii_table[(guchar) (*s)] & G_ASCII_UPPER) != 0) ? *s - 'A' + 'a' : *s;
269
270 return (str);
271}
272
273/* Convert all ASCII letters to upper case, in place. */
274char *
275ascii_strup_inplace(char *str)
276{
277 char *s;
278
279 for (s = str; *s; s++)
280 /* What 'g_ascii_toupper (char c)' does, this should be slightly more efficient */
281 *s = g_ascii_islower (*s)((g_ascii_table[(guchar) (*s)] & G_ASCII_LOWER) != 0) ? *s - 'a' + 'A' : *s;
282
283 return (str);
284}
285
286/* Check if an entire string is printable. */
287bool_Bool
288isprint_string(const char *str)
289{
290 unsigned pos;
291
292 /* Loop until we reach the end of the string (a null) */
293 for(pos = 0; str[pos] != '\0'; pos++){
294 if(!g_ascii_isprint(str[pos])((g_ascii_table[(guchar) (str[pos])] & G_ASCII_PRINT) != 0
)
){
295 /* The string contains a non-printable character */
296 return false0;
297 }
298 }
299
300 /* The string contains only printable characters */
301 return true1;
302}
303
304/* Check if an entire UTF-8 string is printable. */
305bool_Bool
306isprint_utf8_string(const char *str, const unsigned length)
307{
308 const char *strend = str + length;
309
310 if (!g_utf8_validate(str, length, NULL((void*)0))) {
311 return false0;
312 }
313
314 while (str < strend) {
315 /* This returns false for G_UNICODE_CONTROL | G_UNICODE_FORMAT |
316 * G_UNICODE_UNASSIGNED | G_UNICODE_SURROGATE
317 * XXX: Could it be ok to have certain format characters, e.g.
318 * U+00AD SOFT HYPHEN? If so, format_text() should be changed too.
319 */
320 if (!g_unichar_isprint(g_utf8_get_char(str))) {
321 return false0;
322 }
323 str = g_utf8_next_char(str)((str) + g_utf8_skip[*(const guchar *)(str)]);
324 }
325
326 return true1;
327}
328
329/* Check if an entire string is digits. */
330bool_Bool
331isdigit_string(const char *str)
332{
333 unsigned pos;
334
335 /* Loop until we reach the end of the string (a null) */
336 for(pos = 0; str[pos] != '\0'; pos++){
337 if(!g_ascii_isdigit(str[pos])((g_ascii_table[(guchar) (str[pos])] & G_ASCII_DIGIT) != 0
)
){
338 /* The string contains a non-digit character */
339 return false0;
340 }
341 }
342
343 /* The string contains only digits */
344 return true1;
345}
346
347const char *
348ws_ascii_strcasestr(const char *haystack, const char *needle)
349{
350 /* Do not use strcasestr() here, even if a system has it, as it is
351 * locale-dependent (and has different results for e.g. Turkic languages.)
352 * FreeBSD, NetBSD, macOS have a strcasestr_l() that could be used.
353 */
354 size_t hlen = strlen(haystack);
355 size_t nlen = strlen(needle);
356
357 while (hlen-- >= nlen) {
358 if (!g_ascii_strncasecmp(haystack, needle, nlen))
359 return haystack;
360 haystack++;
361 }
362 return NULL((void*)0);
363}
364
365/* Return the last occurrence of ch in the n bytes of haystack.
366 * If not found or n is 0, return NULL. */
367const uint8_t *
368ws_memrchr(const void *_haystack, int ch, size_t n)
369{
370#ifdef HAVE_MEMRCHR1
371 return memrchr(_haystack, ch, n);
372#else
373 /* A generic implementation. This could be optimized considerably,
374 * e.g. by fetching a word at a time.
375 */
376 if (n == 0) {
377 return NULL((void*)0);
378 }
379 const uint8_t *haystack = _haystack;
380 const uint8_t *p;
381 uint8_t c = (uint8_t)ch;
382
383 const uint8_t *const end = haystack + n - 1;
384
385 for (p = end; p >= haystack; --p) {
386 if (*p == c) {
387 return p;
388 }
389 }
390
391 return NULL((void*)0);
392#endif /* HAVE_MEMRCHR */
393}
394
395/* Return the first occurrence of ch in the null-terminated string str.
396 * If not found, return a pointer to the null-terminator. */
397char *
398ws_strchrnul(const char *str, int ch)
399{
400#ifdef HAVE_STRCHRNUL1
401 #ifdef __APPLE__
402 /* strchrnul was introduced in macOS 15.4, runtime check if building
403 * with a newer SDK than that but an older deployment target. */
404 if (__builtin_available(macOS 15.4, *)) {
405 return (char*)strchrnul(str, ch);
406 } else {
407 /* Minimal generic implementation. */
408 while (*str != '\0' && *str != (char)ch) {
409 str++;
410 }
411 return (char *)str;
412 }
413 #else
414 /* Cast in case someone has an implementation that works like one of those
415 * fancy C23 qualifier-preserving versions. */
416 return (char*)strchrnul(str, ch);
417 #endif
418#else
419 /* Minimal generic implementation. */
420 while (*str != '\0' && *str != (char)ch) {
421 str++;
422 }
423 return (char *)str;
424#endif
425}
426
427static const char *thousands_grouping_fmt;
428static const char *thousands_grouping_fmt_flt;
429
430DIAG_OFF(format)clang diagnostic push clang diagnostic ignored "-Wformat"
431static void test_printf_thousands_grouping(void) {
432 /* test whether wmem_strbuf works with "'" flag character */
433 wmem_strbuf_t *buf = wmem_strbuf_new(NULL((void*)0), NULL((void*)0));
434 wmem_strbuf_append_printf(buf, "%'d", 22);
435 if (g_strcmp0(wmem_strbuf_get_str(buf), "22") == 0) {
436 thousands_grouping_fmt = "%'"PRId64"l" "d";
437 thousands_grouping_fmt_flt = "%'.*f";
438 } else {
439 /* Don't use */
440 thousands_grouping_fmt = "%"PRId64"l" "d";
441 thousands_grouping_fmt_flt = "%.*f";
442 }
443 wmem_strbuf_destroy(buf);
444}
445DIAG_ON(format)clang diagnostic pop
446
447static const char* decimal_point = NULL((void*)0);
448
449static void truncate_numeric_strbuf(wmem_strbuf_t *strbuf, int n) {
450
451 const char *s = wmem_strbuf_get_str(strbuf);
452 const char *p;
453 int count;
454
455 if (decimal_point == NULL((void*)0)) {
456 decimal_point = localeconv()->decimal_point;
457 }
458
459 p = strchr(s, decimal_point[0])_Generic (0 ? (s) : (void *) 1, const void *: (const char *) (
strchr (s, decimal_point[0])), default: strchr (s, decimal_point
[0]))
;
460 if (p != NULL((void*)0)) {
461 count = n;
462 while (count >= 0) {
463 count--;
464 if (*p == '\0')
465 break;
466 p++;
467 }
468
469 p--;
470 while (*p == '0') {
471 p--;
472 }
473
474 if (*p != decimal_point[0]) {
475 p++;
476 }
477 wmem_strbuf_truncate(strbuf, (size_t)(p - s));
478 }
479}
480
481/* Given a floating point value, return it in a human-readable format,
482 * using units with metric prefixes (falling back to scientific notation
483 * with the base units if outside the range.)
484 */
485char *
486format_units(wmem_allocator_t *allocator, double size,
487 format_size_units_e unit, uint16_t flags,
488 int precision)
489{
490 wmem_strbuf_t *human_str = wmem_strbuf_new(allocator, NULL((void*)0));
491 bool_Bool is_small = false0;
492 /* is_small is when to use the longer, spelled out unit.
493 * We use it for inf, NaN, 0, and unprefixed small values,
494 * but not for unprefixed values using scientific notation
495 * the value is outside the supported prefix range.
496 */
497 bool_Bool scientific = false0;
498 double abs_size = fabs(size);
499 const struct prefix_parameters * const pp = prefix_parameters_for_flags(flags);
500 int prefix_index = pp->prefix_offset;
501 char *ret_val;
502
503 if (thousands_grouping_fmt == NULL((void*)0))
504 test_printf_thousands_grouping();
505
506 if (isfinite(size)__builtin_isfinite (size) && size != 0.0) {
507
508 double comp = precision == 0 ? 10.0 : 1.0;
509
510 /* For precision 0, use the range [10, 10*power) because only
511 * one significant digit is not as useful. This is what format_size
512 * does for integers. ("ls -h" uses one digit after the decimal
513 * point only for the [1, 10) range, g_format_size() always displays
514 * tenths.) Prefer non-prefixed units for the range [1,10), though.
515 *
516 * We have a limited number of units to check, so this (which
517 * can be unrolled) is presumably faster than log + floor + pow/exp
518 */
519 if (abs_size < 1.0) {
520 while (abs_size < comp) {
521 abs_size *= pp->power;
522 if (prefix_index == 0) {
523 scientific = true1;
524 break;
525 }
526 prefix_index--;
527 }
528 } else {
529 while (abs_size >= comp * pp->power) {
530 abs_size /= pp->power;
531 if (prefix_index == pp->prefix_count - 1) {
532 scientific = true1;
533 break;
534 }
535 prefix_index++;
536 }
537 }
538 }
539
540 if (scientific) {
541 wmem_strbuf_append_printf(human_str, "%.*g", precision + 1, size);
542 prefix_index = pp->prefix_offset;
543 } else {
544 if (prefix_index == pp->prefix_offset) {
545 is_small = true1;
546 }
547 size = copysign(abs_size, size);
548 // Truncate trailing zeros, but do it this way because we know
549 // we don't want scientific notation, and we don't want %g to
550 // switch to that if precision is small. (We could always use
551 // %g when precision is large.)
552 wmem_strbuf_append_printf(human_str, thousands_grouping_fmt_flt, precision, size);
553 truncate_numeric_strbuf(human_str, precision);
554 // XXX - when rounding to a certain precision, printf might
555 // round up to "power" from something like 999.99999995, which
556 // looks a little odd on a graph when transitioning from 1,000 bytes
557 // (for values just under 1 kB) to 1 kB (for values 1 kB and larger.)
558 // Due to edge cases in binary fp representation and how printf might
559 // round things, the right way to handle it is taking the printf output
560 // and comparing it to "1000" and "1024" and adjusting the exponent
561 // if so - though we need to compare to the version with the thousands
562 // separator if we have that (which makes it harder to use strnatcmp
563 // as is.)
564 }
565
566 wmem_strbuf_append(human_str, pp->prefix[prefix_index]);
567
568 switch (unit) {
569 case FORMAT_SIZE_UNIT_NONE:
570 break;
571 case FORMAT_SIZE_UNIT_BYTES:
572 wmem_strbuf_append(human_str, is_small ? "bytes" : "B");
573 break;
574 case FORMAT_SIZE_UNIT_BITS:
575 wmem_strbuf_append(human_str, is_small ? "bits" : "b");
576 break;
577 case FORMAT_SIZE_UNIT_BITS_S:
578 wmem_strbuf_append(human_str, is_small ? "bits/s" : "bps");
579 break;
580 case FORMAT_SIZE_UNIT_BYTES_S:
581 wmem_strbuf_append(human_str, is_small ? "bytes/s" : "Bps");
582 break;
583 case FORMAT_SIZE_UNIT_PACKETS:
584 wmem_strbuf_append(human_str, is_small ? "packets" : "pkts");
585 break;
586 case FORMAT_SIZE_UNIT_PACKETS_S:
587 wmem_strbuf_append(human_str, is_small ? "packets/s" : "pkts/s");
588 break;
589 case FORMAT_SIZE_UNIT_EVENTS:
590 wmem_strbuf_append(human_str, is_small ? "events" : "evts");
591 break;
592 case FORMAT_SIZE_UNIT_EVENTS_S:
593 wmem_strbuf_append(human_str, is_small ? "events/s" : "evts/s");
594 break;
595 case FORMAT_SIZE_UNIT_FIELDS:
596 wmem_strbuf_append(human_str, is_small ? "fields" : "flds");
597 break;
598 case FORMAT_SIZE_UNIT_SECONDS:
599 wmem_strbuf_append(human_str, is_small ? "seconds" : "s");
600 break;
601 case FORMAT_SIZE_UNIT_ERLANGS:
602 wmem_strbuf_append(human_str, is_small ? "erlangs" : "E");
603 break;
604 default:
605 ws_assert_not_reached()ws_log_fatal_full("", LOG_LEVEL_ERROR, "wsutil/str_util.c", 605
, __func__, "assertion \"not reached\" failed")
;
606 }
607
608 ret_val = wmem_strbuf_finalize(human_str);
609 /* Convention is a space between the value and the units. If we have
610 * a prefix, the space is before the prefix. There are two possible
611 * uses of FORMAT_SIZE_UNIT_NONE:
612 * 1. Add a unit immediately after the string returned. In this case,
613 * we would want the string to end with a space if there's no prefix.
614 * 2. The unit appears somewhere else, e.g. in a legend, header, or
615 * different column. In this case, we don't want the string to end
616 * with a space if there's no prefix.
617 * chomping the string here, as we've traditionally done, optimizes for
618 * the latter case but makes the former case harder.
619 * Perhaps the right approach is to distinguish the cases with a new
620 * enum value.
621 */
622 return g_strchomp(ret_val);
623}
624
625/* Given a size, return its value in a human-readable format */
626/* This doesn't handle fractional values. We might want to just
627 * call the version with the double and precision 0 (possibly
628 * slower due to the use of floating point math, but do we care?)
629 */
630char *
631format_size_wmem(wmem_allocator_t *allocator, int64_t size,
632 format_size_units_e unit, uint16_t flags)
633{
634 wmem_strbuf_t *human_str = wmem_strbuf_new(allocator, NULL((void*)0));
635 bool_Bool is_small = false0;
636 const struct prefix_parameters * const pp = prefix_parameters_for_flags(flags);
637 char *ret_val;
638
639 if (thousands_grouping_fmt == NULL((void*)0))
640 test_printf_thousands_grouping();
641
642 int prefix_index = pp->prefix_offset;
643 int64_t scale = 1;
644 while (prefix_index + 1 < pp->prefix_count && scale < INT64_MAX(9223372036854775807L) / (10 * pp->power) && size >= scale * pp->power * 10) {
645 prefix_index++;
646 scale *= pp->power;
647 }
648
649 wmem_strbuf_append_printf(human_str, thousands_grouping_fmt, size / scale);
650 wmem_strbuf_append(human_str, pp->prefix[prefix_index]);
651 is_small = prefix_index == pp->prefix_offset;
652
653 switch (unit) {
654 case FORMAT_SIZE_UNIT_NONE:
655 break;
656 case FORMAT_SIZE_UNIT_BYTES:
657 wmem_strbuf_append(human_str, is_small ? "bytes" : "B");
658 break;
659 case FORMAT_SIZE_UNIT_BITS:
660 wmem_strbuf_append(human_str, is_small ? "bits" : "b");
661 break;
662 case FORMAT_SIZE_UNIT_BITS_S:
663 wmem_strbuf_append(human_str, is_small ? "bits/s" : "bps");
664 break;
665 case FORMAT_SIZE_UNIT_BYTES_S:
666 wmem_strbuf_append(human_str, is_small ? "bytes/s" : "Bps");
667 break;
668 case FORMAT_SIZE_UNIT_PACKETS:
669 wmem_strbuf_append(human_str, is_small ? "packets" : "pkts");
670 break;
671 case FORMAT_SIZE_UNIT_PACKETS_S:
672 wmem_strbuf_append(human_str, is_small ? "packets/s" : "pkts/s");
673 break;
674 case FORMAT_SIZE_UNIT_EVENTS:
675 wmem_strbuf_append(human_str, is_small ? "events" : "evts");
676 break;
677 case FORMAT_SIZE_UNIT_EVENTS_S:
678 wmem_strbuf_append(human_str, is_small ? "events/s" : "evts/s");
679 break;
680 case FORMAT_SIZE_UNIT_FIELDS:
681 wmem_strbuf_append(human_str, is_small ? "fields" : "flds");
682 break;
683 case FORMAT_SIZE_UNIT_SECONDS:
684 wmem_strbuf_append(human_str, is_small ? "seconds" : "s");
685 break;
686 case FORMAT_SIZE_UNIT_ERLANGS:
687 wmem_strbuf_append(human_str, is_small ? "erlangs" : "E");
688 break;
689 default:
690 ws_assert_not_reached()ws_log_fatal_full("", LOG_LEVEL_ERROR, "wsutil/str_util.c", 690
, __func__, "assertion \"not reached\" failed")
;
691 }
692
693 ret_val = wmem_strbuf_finalize(human_str);
694 return g_strchomp(ret_val);
695}
696
697char
698printable_char_or_period(char c)
699{
700 return g_ascii_isprint(c)((g_ascii_table[(guchar) (c)] & G_ASCII_PRINT) != 0) ? c : '.';
701}
702
703/*
704 * This is used by the display filter engine and must be compatible
705 * with display filter syntax.
706 */
707static inline bool_Bool
708escape_char(char c, char *p)
709{
710 int r = -1;
711 ws_assert(p)do { if ((1) && !(p)) ws_log_fatal_full("", LOG_LEVEL_ERROR
, "wsutil/str_util.c", 711, __func__, "assertion failed: %s",
"p"); } while (0)
;
712
713 /*
714 * backslashes and double-quotes must be escaped (double-quotes
715 * are escaped by passing '"' as quote_char in escape_string_len)
716 * whitespace is also escaped.
717 */
718 switch (c) {
719 case '\a': r = 'a'; break;
720 case '\b': r = 'b'; break;
721 case '\f': r = 'f'; break;
722 case '\n': r = 'n'; break;
723 case '\r': r = 'r'; break;
724 case '\t': r = 't'; break;
725 case '\v': r = 'v'; break;
726 case '\\': r = '\\'; break;
727 case '\0': r = '0'; break;
728 }
729
730 if (r != -1) {
731 *p = r;
732 return true1;
733 }
734 return false0;
735}
736
737static inline bool_Bool
738escape_null(char c, char *p)
739{
740 ws_assert(p)do { if ((1) && !(p)) ws_log_fatal_full("", LOG_LEVEL_ERROR
, "wsutil/str_util.c", 740, __func__, "assertion failed: %s",
"p"); } while (0)
;
741 if (c == '\0') {
742 *p = '0';
743 return true1;
744 }
745 return false0;
746}
747
748static char *
749escape_string_len(wmem_allocator_t *alloc, const char *string, ssize_t len,
750 bool_Bool (*escape_func)(char c, char *p), bool_Bool add_quotes,
751 char quote_char, bool_Bool double_quote)
752{
753 char c, r;
754 wmem_strbuf_t *buf;
755 size_t abs_len, alloc_size, i;
756
757 abs_len = (len < 0) ? strlen(string) : (size_t)len;
758
759 alloc_size = abs_len;
760 if (add_quotes)
761 alloc_size += 2;
762
763 buf = wmem_strbuf_new_sized(alloc, alloc_size);
764
765 if (add_quotes && quote_char != '\0')
766 wmem_strbuf_append_c(buf, quote_char);
767
768 for (i = 0; i < abs_len; i++) {
769 c = string[i];
770 if ((escape_func(c, &r))) {
771 wmem_strbuf_append_c(buf, '\\');
772 wmem_strbuf_append_c(buf, r);
773 }
774 else if (c == quote_char && quote_char != '\0') {
775 /* If quoting, we must escape the quote_char somehow. */
776 if (double_quote) {
777 wmem_strbuf_append_c(buf, c);
778 wmem_strbuf_append_c(buf, c);
779 } else {
780 wmem_strbuf_append_c(buf, '\\');
781 wmem_strbuf_append_c(buf, c);
782 }
783 }
784 else if (c == '\\' && quote_char != '\0' && !double_quote) {
785 /* If quoting, and escaping the quote_char with a backslash,
786 * then backslash must be escaped, even if escape_func doesn't. */
787 wmem_strbuf_append_c(buf, '\\');
788 wmem_strbuf_append_c(buf, '\\');
789 }
790 else {
791 /* Other UTF-8 bytes are passed through. */
792 wmem_strbuf_append_c(buf, c);
793 }
794 }
795
796 if (add_quotes && quote_char != '\0')
797 wmem_strbuf_append_c(buf, quote_char);
798
799 return wmem_strbuf_finalize(buf);
800}
801
802char *
803ws_escape_string_len(wmem_allocator_t *alloc, const char *string, ssize_t len, bool_Bool add_quotes)
804{
805 return escape_string_len(alloc, string, len, escape_char, add_quotes, '"', false0);
806}
807
808char *
809ws_escape_string(wmem_allocator_t *alloc, const char *string, bool_Bool add_quotes)
810{
811 return escape_string_len(alloc, string, -1, escape_char, add_quotes, '"', false0);
812}
813
814char *ws_escape_null(wmem_allocator_t *alloc, const char *string, size_t len, bool_Bool add_quotes)
815{
816 /* XXX: The existing behavior (maintained) here is not to escape
817 * backslashes even though NUL is escaped.
818 */
819 return escape_string_len(alloc, string, len, escape_null, add_quotes, add_quotes ? '"' : '\0', false0);
820}
821
822char *ws_escape_csv(wmem_allocator_t *alloc, const char *string, bool_Bool add_quotes, char quote_char, bool_Bool double_quote, bool_Bool escape_whitespace)
823{
824 if (escape_whitespace)
825 return escape_string_len(alloc, string, -1, escape_char, add_quotes, quote_char, double_quote);
826 else
827 return escape_string_len(alloc, string, -1, escape_null, add_quotes, quote_char, double_quote);
828}
829
830bool_Bool ws_csv_value_is_formula(const char *string)
831{
832 if (string == NULL((void*)0))
833 return false0;
834
835 /* Some spreadsheet applications strip leading spaces on import and then
836 * evaluate what follows, so look past them. */
837 while (*string == ' ')
838 string++;
839
840 switch (*string) {
841 case '=': /* Formula */
842 case '+': /* Formula */
843 case '-': /* Formula */
844 case '@': /* Lotus-style function, still honored for compatibility */
845 case '\t': /* Stripped on import, so it can precede any of the above */
846 case '\r':
847 return true1;
848 default:
849 return false0;
850 }
851}
852
853typedef enum {
854 UNESCAPE_NONE,
855 UNESCAPE_ESCAPE,
856} unescape_state_e;
857
858wmem_strbuf_t *ws_unescape_string_len(wmem_allocator_t *alloc, const uint8_t *string, ssize_t len, GError **err)
859{
860 wmem_strbuf_t *buf;
861 size_t abs_len, i;
862 unsigned char c;
863 unescape_state_e state = UNESCAPE_NONE;
864 bool_Bool possibly_invalid = false0;
865 int value;
866 gunichar cp;
867
868 abs_len = (len < 0) ? strlen((const char*)string) : (size_t)len;
869
870 buf = wmem_strbuf_new_sized(alloc, abs_len);
871
872 /* With \u and \U escapes, we can only produce a result by knowing the
873 * intended final encoding. (We assume UTF-8 here.) \x escapes do not
874 * assume the final encoding, but neither do they ensure that the result
875 * will be valid in whatever encoding is used. The other escapes still
876 * assume an ASCII-like encoding for the C0 control characters.
877 *
878 * This attempts to be permissive and allow C, C++, Python, and JSON
879 * escapes, though C and C++ \x escapes longer than two characters are
880 * not allowed, since the target encoding is UTF-8. */
881 for (i = 0; i < abs_len; i++) {
882 c = string[i];
883 switch (state) {
884 case UNESCAPE_NONE:
885 switch (c) {
886 case '\\':
887 state = UNESCAPE_ESCAPE;
888 break;
889 default:
890 if (c & 0x80) {
891 /* XXX - Use ws_utf8_char_len and validate now?
892 * Trickier because there might, somehow, be a single
893 * UTF-8 character which is a mix of escaped and
894 * unescaped bytes. */
895 possibly_invalid = true1;
896 }
897 /* Pass through. */
898 wmem_strbuf_append_c(buf, c);
899 }
900 break;
901 case UNESCAPE_ESCAPE:
902 switch (c) {
903 case 'a':
904 wmem_strbuf_append_c(buf, '\a');
905 break;
906 case 'b':
907 wmem_strbuf_append_c(buf, '\b');
908 break;
909 case 'f':
910 wmem_strbuf_append_c(buf, '\f');
911 break;
912 case 'n':
913 wmem_strbuf_append_c(buf, '\n');
914 break;
915 case 'r':
916 wmem_strbuf_append_c(buf, '\r');
917 break;
918 case 't':
919 wmem_strbuf_append_c(buf, '\t');
920 break;
921 case 'v':
922 wmem_strbuf_append_c(buf, '\v');
923 break;
924 case '\\':
925 wmem_strbuf_append_c(buf, '\\');
926 break;
927 case '\"':
928 wmem_strbuf_append_c(buf, '\"');
929 break;
930 case '\'':
931 // C, C++, Python (not JSON)
932 wmem_strbuf_append_c(buf, '\'');
933 break;
934 case '?':
935 // C and C++ \? for supporting the late unlamented trigraphs
936 wmem_strbuf_append_c(buf, '?');
937 break;
938 case 'x':
939 // C and C++ allow an unlimited number of hex characters,
940 // but the end value must fit in a code unit of the string
941 // type, e.g. 8-bit for UTF-8. (Not a Unicode code point.)
942 // Python requires two characters exactly (which also means
943 // that if the third is valid hex digit that's fine.)
944 // JSON doesn't support.
945 ++i;
946 if (i <= abs_len) {
947 g_set_error(err, G_CONVERT_ERRORg_convert_error_quark(), G_CONVERT_ERROR_PARTIAL_INPUT,
948 "Hexadecimal character missing after \\x");
949 goto out;
950 }
951 value = ws_xton(string[i]);
952 if (value == -1) {
953 g_set_error(err, G_CONVERT_ERRORg_convert_error_quark(), G_CONVERT_ERROR_ILLEGAL_SEQUENCE,
954 "Non-hexadecimal character after \\x");
955 goto out;
956 }
957 if (((abs_len - i) > 1) && ws_xton(string[i+1])) {
958 ++i;
959 value <<= 4;
960 value |= ws_xton(string[i]);
961 }
962 c = (char)value;
963 if (c & 0x80) {
964 possibly_invalid = true1;
965 }
966 wmem_strbuf_append_c(buf, c);
967 break;
968 case 'u':
969 ++i;
970 if ((abs_len - i) < 4) {
971 g_set_error(err, G_CONVERT_ERRORg_convert_error_quark(), G_CONVERT_ERROR_PARTIAL_INPUT,
972 "\\u must be followed by four characters");
973 goto out;
974 }
975 if (!ws_hexbuftou32(&string[i], 4, NULL((void*)0), &cp)) {
976 g_set_error(err, G_CONVERT_ERRORg_convert_error_quark(), G_CONVERT_ERROR_ILLEGAL_SEQUENCE,
977 "\\u must be followed by four hexadecimal characters");
978 goto out;
979 }
980 i += 4;
981 /* JSON (but not C, C++, Python) can represent code points outside
982 * the BMP with UTF-16 surrogate pairs. */
983 if (IS_LEAD_SURROGATE(cp)((cp) >= 0xd800 && (cp) < 0xdc00)) {
984 /* high surrogate */
985 if ((abs_len - i) < 6) {
986 g_set_error(err, G_CONVERT_ERRORg_convert_error_quark(), G_CONVERT_ERROR_PARTIAL_INPUT,
987 "\\u high surrogate must be followed by escaped low surrogate");
988 goto out;
989 }
990 uint16_t second_code;
991 if (string[i] != '\\' || string[i + 1] != 'u' ||
992 !ws_hexbuftou16(&string[i+2], 4, NULL((void*)0), &second_code)) {
993 g_set_error(err, G_CONVERT_ERRORg_convert_error_quark(), G_CONVERT_ERROR_ILLEGAL_SEQUENCE,
994 "\\u high surrogate must be followed by escaped low surrogate");
995 goto out;
996 }
997 i += 6;
998 if (!IS_TRAIL_SURROGATE(second_code)((second_code) >= 0xdc00 && (second_code) < 0xe000
)
) {
999 g_set_error(err, G_CONVERT_ERRORg_convert_error_quark(), G_CONVERT_ERROR_ILLEGAL_SEQUENCE,
1000 "\\u followed by unpaired UTF-16 surrogate");
1001 goto out;
1002 }
1003 cp = SURROGATE_VALUE(cp, second_code)(((((cp) - 0xd800) << 10) | ((second_code) - 0xdc00)) +
0x10000)
;
1004 } else if (IS_TRAIL_SURROGATE(cp)((cp) >= 0xdc00 && (cp) < 0xe000)) {
1005 /* isolated low surrogate, not allowed. */
1006 g_set_error(err, G_CONVERT_ERRORg_convert_error_quark(), G_CONVERT_ERROR_ILLEGAL_SEQUENCE,
1007 "\\u followed by isolated low surrogate");
1008 goto out;
1009 }
1010 wmem_strbuf_append_unichar_validated(buf, cp);
1011 break;
1012 case 'U':
1013 ++i;
1014 if ((abs_len - i) < 8) {
1015 g_set_error(err, G_CONVERT_ERRORg_convert_error_quark(), G_CONVERT_ERROR_PARTIAL_INPUT,
1016 "\\U must be followed by eight characters");
1017 goto out;
1018 }
1019 if (!ws_hexbuftou32(&string[i], 8, NULL((void*)0), &cp)) {
1020 g_set_error(err, G_CONVERT_ERRORg_convert_error_quark(), G_CONVERT_ERROR_ILLEGAL_SEQUENCE,
1021 "\\U must be followed by eight hexadecimal characters");
1022 goto out;
1023 }
1024 wmem_strbuf_append_unichar_validated(buf, cp);
1025 break;
1026 default:
1027 g_set_error(err, G_CONVERT_ERRORg_convert_error_quark(), G_CONVERT_ERROR_ILLEGAL_SEQUENCE,
1028 "Unknown escape sequence");
1029 goto out;
1030 }
1031 state = UNESCAPE_NONE;
1032 break;
1033 default:
1034 /* Pass through. */
1035 wmem_strbuf_append_c(buf, c);
1036 }
1037 }
1038 switch (state) {
1039 case UNESCAPE_ESCAPE:
1040 g_set_error(err, G_CONVERT_ERRORg_convert_error_quark(), G_CONVERT_ERROR_PARTIAL_INPUT,
1041 "\\ at the end of input");
1042 goto out;
1043 default:
1044 break;
1045 }
1046
1047out:
1048 /* Right now on error this passes through what was successfully decoded.
1049 * We could also replace with replacement characters and continue if
1050 * possible, or return a strbuf with an empty string. */
1051 if (possibly_invalid) {
1052 wmem_strbuf_utf8_make_valid(buf);
1053 }
1054 return buf;
1055}
1056
1057const char *
1058ws_strerrorname_r(int errnum, char *buf, size_t buf_size)
1059{
1060#ifdef HAVE_STRERRORNAME_NP1
1061 const char *errstr = strerrorname_np(errnum);
1062 if (errstr != NULL((void*)0)) {
1063 (void)g_strlcpy(buf, errstr, buf_size);
1064 return buf;
1065 }
1066#endif
1067 snprintf(buf, buf_size, "Errno(%d)", errnum);
1068 return buf;
1069}
1070
1071char *
1072ws_strdup_underline(wmem_allocator_t *allocator, long offset, size_t len)
1073{
1074 if (offset < 0)
1075 return NULL((void*)0);
1076
1077 wmem_strbuf_t *buf = wmem_strbuf_new_sized(allocator, offset + len);
1078
1079 for (int i = 0; i < offset; i++) {
1080 wmem_strbuf_append_c(buf, ' ');
1081 }
1082 wmem_strbuf_append_c(buf, '^');
1083
1084 for (size_t l = len; l > 1; l--) {
1085 wmem_strbuf_append_c(buf, '~');
1086 }
1087
1088 return wmem_strbuf_finalize(buf);
1089}
1090
1091#define INITIAL_FMTBUF_SIZE128 128
1092
1093/*
1094 * Declare, and initialize, the variables used for an output buffer.
1095 */
1096#define FMTBUF_VARSchar *fmtbuf = (char*)wmem_alloc(allocator, 128); unsigned fmtbuf_len
= 128; unsigned column = 0
\
1097 char *fmtbuf = (char*)wmem_alloc(allocator, INITIAL_FMTBUF_SIZE128); \
1098 unsigned fmtbuf_len = INITIAL_FMTBUF_SIZE128; \
1099 unsigned column = 0
1100
1101/*
1102 * Expand the buffer to be large enough to add nbytes bytes, plus a
1103 * terminating '\0'.
1104 */
1105#define FMTBUF_EXPAND(nbytes)if (column+(nbytes+1) >= fmtbuf_len) { if (__builtin_mul_overflow
((fmtbuf_len), (2), (&fmtbuf_len))) { do { if (1) { ws_log_full
("", LOG_LEVEL_DEBUG, "wsutil/str_util.c", 1105, __func__, "overflow."
); } } while (0); fmtbuf[column] = '\0'; return fmtbuf; } if (
column+(nbytes+1) >= fmtbuf_len) { if (__builtin_add_overflow
((fmtbuf_len), ((column + nbytes + 2) - fmtbuf_len), (&fmtbuf_len
))) { do { if (1) { ws_log_full("", LOG_LEVEL_DEBUG, "wsutil/str_util.c"
, 1105, __func__, "overflow."); } } while (0); fmtbuf[column]
= '\0'; return fmtbuf; } } fmtbuf = (char *)wmem_realloc(allocator
, fmtbuf, fmtbuf_len); }
\
1106 /* \
1107 * Is there enough room for those bytes and also enough room for \
1108 * a terminating '\0'? \
1109 */ \
1110 if (column+(nbytes+1) >= fmtbuf_len) { \
1111 /* \
1112 * Double the buffer's size if it's not big enough. \
1113 * The size of the buffer starts at 128, so doubling its size \
1114 * adds at least another 128 bytes, which is more than enough \
1115 * for one more character plus a terminating '\0'. \
1116 */ \
1117 if (ckd_mul(&fmtbuf_len, fmtbuf_len, 2)__builtin_mul_overflow((fmtbuf_len), (2), (&fmtbuf_len))) { \
1118 ws_debug("overflow.")do { if (1) { ws_log_full("", LOG_LEVEL_DEBUG, "wsutil/str_util.c"
, 1118, __func__, "overflow."); } } while (0)
; \
1119 FMTBUF_ENDSTRfmtbuf[column] = '\0'; \
1120 return fmtbuf; \
1121 } \
1122 if (column+(nbytes+1) >= fmtbuf_len) { \
1123 if (ckd_add(&fmtbuf_len, fmtbuf_len, (column + nbytes + 2) - fmtbuf_len)__builtin_add_overflow((fmtbuf_len), ((column + nbytes + 2) -
fmtbuf_len), (&fmtbuf_len))
) { \
1124 ws_debug("overflow.")do { if (1) { ws_log_full("", LOG_LEVEL_DEBUG, "wsutil/str_util.c"
, 1124, __func__, "overflow."); } } while (0)
; \
1125 FMTBUF_ENDSTRfmtbuf[column] = '\0'; \
1126 return fmtbuf; \
1127 } \
1128 } \
1129 fmtbuf = (char *)wmem_realloc(allocator, fmtbuf, fmtbuf_len); \
1130 }
1131
1132/*
1133 * Put a byte into the buffer; space must have been ensured for it.
1134 */
1135#define FMTBUF_PUTCHAR(b)fmtbuf[column] = (b); column++ \
1136 fmtbuf[column] = (b); \
1137 column++
1138
1139/*
1140 * Add the one-byte argument, as an octal escape sequence, to the end
1141 * of the buffer.
1142 */
1143#define FMTBUF_PUTBYTE_OCTAL(b)fmtbuf[column] = ((((b)>>6)&03) + '0'); column++; fmtbuf
[column] = ((((b)>>3)&07) + '0'); column++; fmtbuf[
column] = ((((b)>>0)&07) + '0'); column++
\
1144 FMTBUF_PUTCHAR((((b)>>6)&03) + '0')fmtbuf[column] = ((((b)>>6)&03) + '0'); column++; \
1145 FMTBUF_PUTCHAR((((b)>>3)&07) + '0')fmtbuf[column] = ((((b)>>3)&07) + '0'); column++; \
1146 FMTBUF_PUTCHAR((((b)>>0)&07) + '0')fmtbuf[column] = ((((b)>>0)&07) + '0'); column++
1147
1148/*
1149 * Add the one-byte argument, as a hex escape sequence, to the end
1150 * of the buffer.
1151 */
1152#define FMTBUF_PUTBYTE_HEX(b)fmtbuf[column] = ('\\'); column++; fmtbuf[column] = ('x'); column
++; fmtbuf[column] = (hex[((b) >> 4) & 0xF]); column
++; fmtbuf[column] = (hex[((b) >> 0) & 0xF]); column
++
\
1153 FMTBUF_PUTCHAR('\\')fmtbuf[column] = ('\\'); column++; \
1154 FMTBUF_PUTCHAR('x')fmtbuf[column] = ('x'); column++; \
1155 FMTBUF_PUTCHAR(hex[((b) >> 4) & 0xF])fmtbuf[column] = (hex[((b) >> 4) & 0xF]); column++; \
1156 FMTBUF_PUTCHAR(hex[((b) >> 0) & 0xF])fmtbuf[column] = (hex[((b) >> 0) & 0xF]); column++
1157
1158#define FMTBUF_PUTBYTES(bytes, len)if (column+(len+1) >= fmtbuf_len) { if (__builtin_mul_overflow
((fmtbuf_len), (2), (&fmtbuf_len))) { do { if (1) { ws_log_full
("", LOG_LEVEL_DEBUG, "wsutil/str_util.c", 1158, __func__, "overflow."
); } } while (0); fmtbuf[column] = '\0'; return fmtbuf; } if (
column+(len+1) >= fmtbuf_len) { if (__builtin_add_overflow
((fmtbuf_len), ((column + len + 2) - fmtbuf_len), (&fmtbuf_len
))) { do { if (1) { ws_log_full("", LOG_LEVEL_DEBUG, "wsutil/str_util.c"
, 1158, __func__, "overflow."); } } while (0); fmtbuf[column]
= '\0'; return fmtbuf; } } fmtbuf = (char *)wmem_realloc(allocator
, fmtbuf, fmtbuf_len); } memcpy(&fmtbuf[column], bytes, len
); column += (unsigned)len;
\
1159 FMTBUF_EXPAND(len)if (column+(len+1) >= fmtbuf_len) { if (__builtin_mul_overflow
((fmtbuf_len), (2), (&fmtbuf_len))) { do { if (1) { ws_log_full
("", LOG_LEVEL_DEBUG, "wsutil/str_util.c", 1159, __func__, "overflow."
); } } while (0); fmtbuf[column] = '\0'; return fmtbuf; } if (
column+(len+1) >= fmtbuf_len) { if (__builtin_add_overflow
((fmtbuf_len), ((column + len + 2) - fmtbuf_len), (&fmtbuf_len
))) { do { if (1) { ws_log_full("", LOG_LEVEL_DEBUG, "wsutil/str_util.c"
, 1159, __func__, "overflow."); } } while (0); fmtbuf[column]
= '\0'; return fmtbuf; } } fmtbuf = (char *)wmem_realloc(allocator
, fmtbuf, fmtbuf_len); }
\
1160 memcpy(&fmtbuf[column], bytes, len); \
1161 column += (unsigned)len; // FMTBUF_EXPAND checks for overflow
1162
1163/*
1164 * Put the trailing '\0' at the end of the buffer.
1165 */
1166#define FMTBUF_ENDSTRfmtbuf[column] = '\0' \
1167 fmtbuf[column] = '\0'
1168
1169static char *
1170format_text_internal(wmem_allocator_t *allocator,
1171 const unsigned char *string, size_t len,
1172 bool_Bool replace_space)
1173{
1174 FMTBUF_VARSchar *fmtbuf = (char*)wmem_alloc(allocator, 128); unsigned fmtbuf_len
= 128; unsigned column = 0
;
1175 const unsigned char *prev = string;
1176 const unsigned char *stringend = string + len;
1177 unsigned char c;
1178 size_t printable_bytes = 0;
1179
1180 while (string < stringend) {
1181 /*
1182 * Get the first byte of this character.
1183 */
1184 c = *string++;
1185 if ((0x20 <= c) && (c < 0x7F)) {
1186 /*
1187 * Printable ASCII, so not part of a multi-byte UTF-8 sequence.
1188 * Make sure there's enough room for one more byte, and add
1189 * the character.
1190 */
1191 printable_bytes++;
1192 } else {
1193 if (printable_bytes) {
1194 FMTBUF_PUTBYTES(prev, printable_bytes)if (column+(printable_bytes+1) >= fmtbuf_len) { if (__builtin_mul_overflow
((fmtbuf_len), (2), (&fmtbuf_len))) { do { if (1) { ws_log_full
("", LOG_LEVEL_DEBUG, "wsutil/str_util.c", 1194, __func__, "overflow."
); } } while (0); fmtbuf[column] = '\0'; return fmtbuf; } if (
column+(printable_bytes+1) >= fmtbuf_len) { if (__builtin_add_overflow
((fmtbuf_len), ((column + printable_bytes + 2) - fmtbuf_len),
(&fmtbuf_len))) { do { if (1) { ws_log_full("", LOG_LEVEL_DEBUG
, "wsutil/str_util.c", 1194, __func__, "overflow."); } } while
(0); fmtbuf[column] = '\0'; return fmtbuf; } } fmtbuf = (char
*)wmem_realloc(allocator, fmtbuf, fmtbuf_len); } memcpy(&
fmtbuf[column], prev, printable_bytes); column += (unsigned)printable_bytes
;
;
1195 printable_bytes = 0;
1196 }
1197 if (replace_space && g_ascii_isspace(c)((g_ascii_table[(guchar) (c)] & G_ASCII_SPACE) != 0)) {
1198 /*
1199 * ASCII, so not part of a multi-byte UTF-8 sequence, but
1200 * not printable, but is a space character; show it as a
1201 * blank.
1202 *
1203 * Make sure there's enough room for one more byte, and add
1204 * the blank.
1205 */
1206 FMTBUF_EXPAND(1)if (column+(1 +1) >= fmtbuf_len) { if (__builtin_mul_overflow
((fmtbuf_len), (2), (&fmtbuf_len))) { do { if (1) { ws_log_full
("", LOG_LEVEL_DEBUG, "wsutil/str_util.c", 1206, __func__, "overflow."
); } } while (0); fmtbuf[column] = '\0'; return fmtbuf; } if (
column+(1 +1) >= fmtbuf_len) { if (__builtin_add_overflow(
(fmtbuf_len), ((column + 1 + 2) - fmtbuf_len), (&fmtbuf_len
))) { do { if (1) { ws_log_full("", LOG_LEVEL_DEBUG, "wsutil/str_util.c"
, 1206, __func__, "overflow."); } } while (0); fmtbuf[column]
= '\0'; return fmtbuf; } } fmtbuf = (char *)wmem_realloc(allocator
, fmtbuf, fmtbuf_len); }
;
1207 FMTBUF_PUTCHAR(' ')fmtbuf[column] = (' '); column++;
1208 } else if (c < 128) {
1209 /*
1210 * ASCII, so not part of a multi-byte UTF-8 sequence, but not
1211 * printable.
1212 *
1213 * That requires a minimum of 2 bytes, one for the backslash
1214 * and one for a letter, so make sure we have enough room
1215 * for that, plus a trailing '\0'.
1216 */
1217 FMTBUF_EXPAND(2)if (column+(2 +1) >= fmtbuf_len) { if (__builtin_mul_overflow
((fmtbuf_len), (2), (&fmtbuf_len))) { do { if (1) { ws_log_full
("", LOG_LEVEL_DEBUG, "wsutil/str_util.c", 1217, __func__, "overflow."
); } } while (0); fmtbuf[column] = '\0'; return fmtbuf; } if (
column+(2 +1) >= fmtbuf_len) { if (__builtin_add_overflow(
(fmtbuf_len), ((column + 2 + 2) - fmtbuf_len), (&fmtbuf_len
))) { do { if (1) { ws_log_full("", LOG_LEVEL_DEBUG, "wsutil/str_util.c"
, 1217, __func__, "overflow."); } } while (0); fmtbuf[column]
= '\0'; return fmtbuf; } } fmtbuf = (char *)wmem_realloc(allocator
, fmtbuf, fmtbuf_len); }
;
1218 FMTBUF_PUTCHAR('\\')fmtbuf[column] = ('\\'); column++;
1219 switch (c) {
1220
1221 case '\a':
1222 FMTBUF_PUTCHAR('a')fmtbuf[column] = ('a'); column++;
1223 break;
1224
1225 case '\b':
1226 FMTBUF_PUTCHAR('b')fmtbuf[column] = ('b'); column++; /* BS */
1227 break;
1228
1229 case '\f':
1230 FMTBUF_PUTCHAR('f')fmtbuf[column] = ('f'); column++; /* FF */
1231 break;
1232
1233 case '\n':
1234 FMTBUF_PUTCHAR('n')fmtbuf[column] = ('n'); column++; /* NL */
1235 break;
1236
1237 case '\r':
1238 FMTBUF_PUTCHAR('r')fmtbuf[column] = ('r'); column++; /* CR */
1239 break;
1240
1241 case '\t':
1242 FMTBUF_PUTCHAR('t')fmtbuf[column] = ('t'); column++; /* tab */
1243 break;
1244
1245 case '\v':
1246 FMTBUF_PUTCHAR('v')fmtbuf[column] = ('v'); column++;
1247 break;
1248
1249 default:
1250 /*
1251 * We've already put the backslash, but this
1252 * will put 3 more characters for the octal
1253 * number; make sure we have enough room for
1254 * that, plus the trailing '\0'.
1255 */
1256 FMTBUF_EXPAND(3)if (column+(3 +1) >= fmtbuf_len) { if (__builtin_mul_overflow
((fmtbuf_len), (2), (&fmtbuf_len))) { do { if (1) { ws_log_full
("", LOG_LEVEL_DEBUG, "wsutil/str_util.c", 1256, __func__, "overflow."
); } } while (0); fmtbuf[column] = '\0'; return fmtbuf; } if (
column+(3 +1) >= fmtbuf_len) { if (__builtin_add_overflow(
(fmtbuf_len), ((column + 3 + 2) - fmtbuf_len), (&fmtbuf_len
))) { do { if (1) { ws_log_full("", LOG_LEVEL_DEBUG, "wsutil/str_util.c"
, 1256, __func__, "overflow."); } } while (0); fmtbuf[column]
= '\0'; return fmtbuf; } } fmtbuf = (char *)wmem_realloc(allocator
, fmtbuf, fmtbuf_len); }
;
1257 FMTBUF_PUTBYTE_OCTAL(c)fmtbuf[column] = ((((c)>>6)&03) + '0'); column++; fmtbuf
[column] = ((((c)>>3)&07) + '0'); column++; fmtbuf[
column] = ((((c)>>0)&07) + '0'); column++
;
1258 break;
1259 }
1260 } else {
1261 /*
1262 * We've fetched the first byte of a multi-byte UTF-8
1263 * sequence into c.
1264 */
1265 int utf8_len;
1266 unsigned char mask;
1267 gunichar uc;
1268 unsigned char first;
1269
1270 if ((c & 0xe0) == 0xc0) {
1271 /* Starts a 2-byte UTF-8 sequence; 1 byte left */
1272 utf8_len = 1;
1273 mask = 0x1f;
1274 } else if ((c & 0xf0) == 0xe0) {
1275 /* Starts a 3-byte UTF-8 sequence; 2 bytes left */
1276 utf8_len = 2;
1277 mask = 0x0f;
1278 } else if ((c & 0xf8) == 0xf0) {
1279 /* Starts a 4-byte UTF-8 sequence; 3 bytes left */
1280 utf8_len = 3;
1281 mask = 0x07;
1282 } else if ((c & 0xfc) == 0xf8) {
1283 /* Starts an old-style 5-byte UTF-8 sequence; 4 bytes left */
1284 utf8_len = 4;
1285 mask = 0x03;
1286 } else if ((c & 0xfe) == 0xfc) {
1287 /* Starts an old-style 6-byte UTF-8 sequence; 5 bytes left */
1288 utf8_len = 5;
1289 mask = 0x01;
1290 } else {
1291 /* 0xfe or 0xff or a continuation byte - not valid */
1292 utf8_len = -1;
1293 }
1294 if (utf8_len > 0) {
1295 /* Try to construct the Unicode character */
1296 uc = c & mask;
1297 for (int i = 0; i < utf8_len; i++) {
1298 if (string >= stringend) {
1299 /*
1300 * Ran out of octets, so the character is
1301 * incomplete. Put in a REPLACEMENT CHARACTER
1302 * instead, and then continue the loop, which
1303 * will terminate.
1304 */
1305 uc = UNICODE_REPLACEMENT_CHARACTER0x00FFFD;
1306 break;
1307 }
1308 c = *string;
1309 if ((c & 0xc0) != 0x80) {
1310 /*
1311 * Not valid UTF-8 continuation character; put in
1312 * a replacement character, and then re-process
1313 * this octet as the beginning of a new character.
1314 */
1315 uc = UNICODE_REPLACEMENT_CHARACTER0x00FFFD;
1316 break;
1317 }
1318 string++;
1319 uc = (uc << 6) | (c & 0x3f);
1320 }
1321
1322 /*
1323 * If this isn't a valid Unicode character, put in
1324 * a REPLACEMENT CHARACTER.
1325 */
1326 if (!g_unichar_validate(uc))
1327 uc = UNICODE_REPLACEMENT_CHARACTER0x00FFFD;
1328 } else {
1329 /* 0xfe or 0xff; put it a REPLACEMENT CHARACTER */
1330 uc = UNICODE_REPLACEMENT_CHARACTER0x00FFFD;
1331 }
1332
1333 /*
1334 * OK, is it a printable Unicode character?
1335 */
1336 if (g_unichar_isprint(uc)) {
1337 /*
1338 * Yes - put it into the string as UTF-8.
1339 * This means that if it was an overlong
1340 * encoding, this will put out the right
1341 * sized encoding.
1342 */
1343 if (uc < 0x80) {
1344 first = 0;
1345 utf8_len = 1;
1346 } else if (uc < 0x800) {
1347 first = 0xc0;
1348 utf8_len = 2;
1349 } else if (uc < 0x10000) {
1350 first = 0xe0;
1351 utf8_len = 3;
1352 } else if (uc < 0x200000) {
1353 first = 0xf0;
1354 utf8_len = 4;
1355 } else if (uc < 0x4000000) {
1356 /*
1357 * This should never happen, as Unicode doesn't
1358 * go that high.
1359 */
1360 first = 0xf8;
1361 utf8_len = 5;
1362 } else {
1363 /*
1364 * This should never happen, as Unicode doesn't
1365 * go that high.
1366 */
1367 first = 0xfc;
1368 utf8_len = 6;
1369 }
1370 FMTBUF_EXPAND(utf8_len)if (column+(utf8_len+1) >= fmtbuf_len) { if (__builtin_mul_overflow
((fmtbuf_len), (2), (&fmtbuf_len))) { do { if (1) { ws_log_full
("", LOG_LEVEL_DEBUG, "wsutil/str_util.c", 1370, __func__, "overflow."
); } } while (0); fmtbuf[column] = '\0'; return fmtbuf; } if (
column+(utf8_len+1) >= fmtbuf_len) { if (__builtin_add_overflow
((fmtbuf_len), ((column + utf8_len + 2) - fmtbuf_len), (&
fmtbuf_len))) { do { if (1) { ws_log_full("", LOG_LEVEL_DEBUG
, "wsutil/str_util.c", 1370, __func__, "overflow."); } } while
(0); fmtbuf[column] = '\0'; return fmtbuf; } } fmtbuf = (char
*)wmem_realloc(allocator, fmtbuf, fmtbuf_len); }
;
1371 for (int i = utf8_len - 1; i > 0; i--) {
1372 fmtbuf[column + i] = (uc & 0x3f) | 0x80;
1373 uc >>= 6;
1374 }
1375 fmtbuf[column] = uc | first;
1376 column += utf8_len;
1377 } else if (replace_space && g_unichar_isspace(uc)) {
1378 /*
1379 * Not printable, but is a space character; show it
1380 * as a blank.
1381 *
1382 * Make sure there's enough room for one more byte,
1383 * and add the blank.
1384 */
1385 FMTBUF_EXPAND(1)if (column+(1 +1) >= fmtbuf_len) { if (__builtin_mul_overflow
((fmtbuf_len), (2), (&fmtbuf_len))) { do { if (1) { ws_log_full
("", LOG_LEVEL_DEBUG, "wsutil/str_util.c", 1385, __func__, "overflow."
); } } while (0); fmtbuf[column] = '\0'; return fmtbuf; } if (
column+(1 +1) >= fmtbuf_len) { if (__builtin_add_overflow(
(fmtbuf_len), ((column + 1 + 2) - fmtbuf_len), (&fmtbuf_len
))) { do { if (1) { ws_log_full("", LOG_LEVEL_DEBUG, "wsutil/str_util.c"
, 1385, __func__, "overflow."); } } while (0); fmtbuf[column]
= '\0'; return fmtbuf; } } fmtbuf = (char *)wmem_realloc(allocator
, fmtbuf, fmtbuf_len); }
;
1386 FMTBUF_PUTCHAR(' ')fmtbuf[column] = (' '); column++;
1387 } else if (c < 128) {
1388 /*
1389 * ASCII, but not printable.
1390 * Yes, this could happen with an overlong encoding.
1391 *
1392 * That requires a minimum of 2 bytes, one for the
1393 * backslash and one for a letter, so make sure we
1394 * have enough room for that, plus a trailing '\0'.
1395 */
1396 FMTBUF_EXPAND(2)if (column+(2 +1) >= fmtbuf_len) { if (__builtin_mul_overflow
((fmtbuf_len), (2), (&fmtbuf_len))) { do { if (1) { ws_log_full
("", LOG_LEVEL_DEBUG, "wsutil/str_util.c", 1396, __func__, "overflow."
); } } while (0); fmtbuf[column] = '\0'; return fmtbuf; } if (
column+(2 +1) >= fmtbuf_len) { if (__builtin_add_overflow(
(fmtbuf_len), ((column + 2 + 2) - fmtbuf_len), (&fmtbuf_len
))) { do { if (1) { ws_log_full("", LOG_LEVEL_DEBUG, "wsutil/str_util.c"
, 1396, __func__, "overflow."); } } while (0); fmtbuf[column]
= '\0'; return fmtbuf; } } fmtbuf = (char *)wmem_realloc(allocator
, fmtbuf, fmtbuf_len); }
;
1397 FMTBUF_PUTCHAR('\\')fmtbuf[column] = ('\\'); column++;
1398 switch (c) {
1399
1400 case '\a':
1401 FMTBUF_PUTCHAR('a')fmtbuf[column] = ('a'); column++;
1402 break;
1403
1404 case '\b':
1405 FMTBUF_PUTCHAR('b')fmtbuf[column] = ('b'); column++; /* BS */
1406 break;
1407
1408 case '\f':
1409 FMTBUF_PUTCHAR('f')fmtbuf[column] = ('f'); column++; /* FF */
1410 break;
1411
1412 case '\n':
1413 FMTBUF_PUTCHAR('n')fmtbuf[column] = ('n'); column++; /* NL */
1414 break;
1415
1416 case '\r':
1417 FMTBUF_PUTCHAR('r')fmtbuf[column] = ('r'); column++; /* CR */
1418 break;
1419
1420 case '\t':
1421 FMTBUF_PUTCHAR('t')fmtbuf[column] = ('t'); column++; /* tab */
1422 break;
1423
1424 case '\v':
1425 FMTBUF_PUTCHAR('v')fmtbuf[column] = ('v'); column++;
1426 break;
1427
1428 default:
1429 /*
1430 * We've already put the backslash, but this
1431 * will put 3 more characters for the octal
1432 * number; make sure we have enough room for
1433 * that, plus the trailing '\0'.
1434 */
1435 FMTBUF_EXPAND(3)if (column+(3 +1) >= fmtbuf_len) { if (__builtin_mul_overflow
((fmtbuf_len), (2), (&fmtbuf_len))) { do { if (1) { ws_log_full
("", LOG_LEVEL_DEBUG, "wsutil/str_util.c", 1435, __func__, "overflow."
); } } while (0); fmtbuf[column] = '\0'; return fmtbuf; } if (
column+(3 +1) >= fmtbuf_len) { if (__builtin_add_overflow(
(fmtbuf_len), ((column + 3 + 2) - fmtbuf_len), (&fmtbuf_len
))) { do { if (1) { ws_log_full("", LOG_LEVEL_DEBUG, "wsutil/str_util.c"
, 1435, __func__, "overflow."); } } while (0); fmtbuf[column]
= '\0'; return fmtbuf; } } fmtbuf = (char *)wmem_realloc(allocator
, fmtbuf, fmtbuf_len); }
;
1436 FMTBUF_PUTBYTE_OCTAL(c)fmtbuf[column] = ((((c)>>6)&03) + '0'); column++; fmtbuf
[column] = ((((c)>>3)&07) + '0'); column++; fmtbuf[
column] = ((((c)>>0)&07) + '0'); column++
;
1437 break;
1438 }
1439 } else {
1440 /*
1441 * Unicode, but not printable, and not ASCII;
1442 * put it out as \uxxxx or \Uxxxxxxxx.
1443 */
1444 if (uc <= 0xFFFF) {
1445 FMTBUF_EXPAND(6)if (column+(6 +1) >= fmtbuf_len) { if (__builtin_mul_overflow
((fmtbuf_len), (2), (&fmtbuf_len))) { do { if (1) { ws_log_full
("", LOG_LEVEL_DEBUG, "wsutil/str_util.c", 1445, __func__, "overflow."
); } } while (0); fmtbuf[column] = '\0'; return fmtbuf; } if (
column+(6 +1) >= fmtbuf_len) { if (__builtin_add_overflow(
(fmtbuf_len), ((column + 6 + 2) - fmtbuf_len), (&fmtbuf_len
))) { do { if (1) { ws_log_full("", LOG_LEVEL_DEBUG, "wsutil/str_util.c"
, 1445, __func__, "overflow."); } } while (0); fmtbuf[column]
= '\0'; return fmtbuf; } } fmtbuf = (char *)wmem_realloc(allocator
, fmtbuf, fmtbuf_len); }
;
1446 FMTBUF_PUTCHAR('\\')fmtbuf[column] = ('\\'); column++;
1447 FMTBUF_PUTCHAR('u')fmtbuf[column] = ('u'); column++;
1448 FMTBUF_PUTCHAR(hex[(uc >> 12) & 0xF])fmtbuf[column] = (hex[(uc >> 12) & 0xF]); column++;
1449 FMTBUF_PUTCHAR(hex[(uc >> 8) & 0xF])fmtbuf[column] = (hex[(uc >> 8) & 0xF]); column++;
1450 FMTBUF_PUTCHAR(hex[(uc >> 4) & 0xF])fmtbuf[column] = (hex[(uc >> 4) & 0xF]); column++;
1451 FMTBUF_PUTCHAR(hex[(uc >> 0) & 0xF])fmtbuf[column] = (hex[(uc >> 0) & 0xF]); column++;
1452 } else {
1453 FMTBUF_EXPAND(10)if (column+(10 +1) >= fmtbuf_len) { if (__builtin_mul_overflow
((fmtbuf_len), (2), (&fmtbuf_len))) { do { if (1) { ws_log_full
("", LOG_LEVEL_DEBUG, "wsutil/str_util.c", 1453, __func__, "overflow."
); } } while (0); fmtbuf[column] = '\0'; return fmtbuf; } if (
column+(10 +1) >= fmtbuf_len) { if (__builtin_add_overflow
((fmtbuf_len), ((column + 10 + 2) - fmtbuf_len), (&fmtbuf_len
))) { do { if (1) { ws_log_full("", LOG_LEVEL_DEBUG, "wsutil/str_util.c"
, 1453, __func__, "overflow."); } } while (0); fmtbuf[column]
= '\0'; return fmtbuf; } } fmtbuf = (char *)wmem_realloc(allocator
, fmtbuf, fmtbuf_len); }
;
1454 FMTBUF_PUTCHAR('\\')fmtbuf[column] = ('\\'); column++;
1455 FMTBUF_PUTCHAR('U')fmtbuf[column] = ('U'); column++;
1456 FMTBUF_PUTCHAR(hex[(uc >> 28) & 0xF])fmtbuf[column] = (hex[(uc >> 28) & 0xF]); column++;
1457 FMTBUF_PUTCHAR(hex[(uc >> 24) & 0xF])fmtbuf[column] = (hex[(uc >> 24) & 0xF]); column++;
1458 FMTBUF_PUTCHAR(hex[(uc >> 20) & 0xF])fmtbuf[column] = (hex[(uc >> 20) & 0xF]); column++;
1459 FMTBUF_PUTCHAR(hex[(uc >> 16) & 0xF])fmtbuf[column] = (hex[(uc >> 16) & 0xF]); column++;
1460 FMTBUF_PUTCHAR(hex[(uc >> 12) & 0xF])fmtbuf[column] = (hex[(uc >> 12) & 0xF]); column++;
1461 FMTBUF_PUTCHAR(hex[(uc >> 8) & 0xF])fmtbuf[column] = (hex[(uc >> 8) & 0xF]); column++;
1462 FMTBUF_PUTCHAR(hex[(uc >> 4) & 0xF])fmtbuf[column] = (hex[(uc >> 4) & 0xF]); column++;
1463 FMTBUF_PUTCHAR(hex[(uc >> 0) & 0xF])fmtbuf[column] = (hex[(uc >> 0) & 0xF]); column++;
1464 }
1465 }
1466 }
1467 prev = string;
1468 }
1469 }
1470 if (printable_bytes) {
1471 FMTBUF_PUTBYTES(prev, printable_bytes)if (column+(printable_bytes+1) >= fmtbuf_len) { if (__builtin_mul_overflow
((fmtbuf_len), (2), (&fmtbuf_len))) { do { if (1) { ws_log_full
("", LOG_LEVEL_DEBUG, "wsutil/str_util.c", 1471, __func__, "overflow."
); } } while (0); fmtbuf[column] = '\0'; return fmtbuf; } if (
column+(printable_bytes+1) >= fmtbuf_len) { if (__builtin_add_overflow
((fmtbuf_len), ((column + printable_bytes + 2) - fmtbuf_len),
(&fmtbuf_len))) { do { if (1) { ws_log_full("", LOG_LEVEL_DEBUG
, "wsutil/str_util.c", 1471, __func__, "overflow."); } } while
(0); fmtbuf[column] = '\0'; return fmtbuf; } } fmtbuf = (char
*)wmem_realloc(allocator, fmtbuf, fmtbuf_len); } memcpy(&
fmtbuf[column], prev, printable_bytes); column += (unsigned)printable_bytes
;
;
1472 printable_bytes = 0;
Value stored to 'printable_bytes' is never read
1473 }
1474
1475 FMTBUF_ENDSTRfmtbuf[column] = '\0';
1476
1477 return fmtbuf;
1478}
1479
1480/*
1481 * Given a wmem scope, a not-necessarily-null-terminated string,
1482 * expected to be in UTF-8 but possibly containing invalid sequences
1483 * (as it may have come from packet data), and the length of the string,
1484 * generate a valid UTF-8 string from it, allocated in the specified
1485 * wmem scope, that:
1486 *
1487 * shows printable Unicode characters as themselves;
1488 *
1489 * shows non-printable ASCII characters as C-style escapes (octal
1490 * if not one of the standard ones such as LF -> '\n');
1491 *
1492 * shows non-printable Unicode-but-not-ASCII characters as
1493 * their universal character names;
1494 *
1495 * shows illegal UTF-8 sequences as a sequence of bytes represented
1496 * as C-style hex escapes (XXX: Does not actually do this. Some illegal
1497 * sequences, such as overlong encodings, the sequences reserved for
1498 * UTF-16 surrogate halves (paired or unpaired), and values outside
1499 * Unicode (i.e., the old sequences for code points above U+10FFFF)
1500 * will be decoded in a permissive way. Other illegal sequences,
1501 * such 0xFE and 0xFF and the presence of a continuation byte where
1502 * not expected (or vice versa its absence), are replaced with
1503 * REPLACEMENT CHARACTER.)
1504 *
1505 * and return a pointer to it.
1506 */
1507char *
1508format_text(wmem_allocator_t *allocator,
1509 const char *string, size_t len)
1510{
1511 return format_text_internal(allocator, (const uint8_t*)string, len, false0);
1512}
1513
1514/** Given a wmem scope and a null-terminated string, expected to be in
1515 * UTF-8 but possibly containing invalid sequences (as it may have come
1516 * from packet data), and the length of the string, generate a valid
1517 * UTF-8 string from it, allocated in the specified wmem scope, that:
1518 *
1519 * shows printable Unicode characters as themselves;
1520 *
1521 * shows non-printable ASCII characters as C-style escapes (octal
1522 * if not one of the standard ones such as LF -> '\n');
1523 *
1524 * shows non-printable Unicode-but-not-ASCII characters as
1525 * their universal character names;
1526 *
1527 * shows illegal UTF-8 sequences as a sequence of bytes represented
1528 * as C-style hex escapes;
1529 *
1530 * and return a pointer to it.
1531 */
1532char *
1533format_text_string(wmem_allocator_t* allocator, const char *string)
1534{
1535 return format_text_internal(allocator, (const uint8_t*)string, strlen(string), false0);
1536}
1537
1538/*
1539 * Given a string, generate a string from it that shows non-printable
1540 * characters as C-style escapes except a whitespace character
1541 * (space, tab, carriage return, new line, vertical tab, or formfeed)
1542 * which will be replaced by a space, and return a pointer to it.
1543 */
1544char *
1545format_text_wsp(wmem_allocator_t* allocator, const char *string, size_t len)
1546{
1547 return format_text_internal(allocator, (const uint8_t*)string, len, true1);
1548}
1549
1550/*
1551 * Given a string, generate a string from it that shows non-printable
1552 * characters as the chr parameter passed, except a whitespace character
1553 * (space, tab, carriage return, new line, vertical tab, or formfeed)
1554 * which will be replaced by a space, and return a pointer to it.
1555 *
1556 * This does *not* treat the input string as UTF-8.
1557 *
1558 * This is useful for displaying binary data that frequently but not always
1559 * contains text; otherwise the number of C escape codes makes it unreadable.
1560 */
1561char *
1562format_text_chr(wmem_allocator_t *allocator, const char *string, size_t len, char chr)
1563{
1564 wmem_strbuf_t *buf;
1565
1566 buf = wmem_strbuf_new_sized(allocator, len + 1);
1567 for (const char *p = string; p < string + len; p++) {
1568 if (g_ascii_isprint(*p)((g_ascii_table[(guchar) (*p)] & G_ASCII_PRINT) != 0)) {
1569 wmem_strbuf_append_c(buf, *p);
1570 }
1571 else if (g_ascii_isspace(*p)((g_ascii_table[(guchar) (*p)] & G_ASCII_SPACE) != 0)) {
1572 wmem_strbuf_append_c(buf, ' ');
1573 }
1574 else {
1575 wmem_strbuf_append_c(buf, chr);
1576 }
1577 }
1578 return wmem_strbuf_finalize(buf);
1579}
1580
1581char *
1582format_char(wmem_allocator_t *allocator, char c)
1583{
1584 char *buf;
1585 char r;
1586
1587 if (g_ascii_isprint(c)((g_ascii_table[(guchar) (c)] & G_ASCII_PRINT) != 0)) {
1588 buf = wmem_alloc_array(allocator, char, 2)((char*)wmem_alloc((allocator), (((((2)) <= 0) || ((size_t
)sizeof(char) > (9223372036854775807L / (size_t)((2))))) ?
0 : (sizeof(char) * ((2))))))
;
1589 buf[0] = c;
1590 buf[1] = '\0';
1591 return buf;
1592 }
1593 if (escape_char(c, &r)) {
1594 buf = wmem_alloc_array(allocator, char, 3)((char*)wmem_alloc((allocator), (((((3)) <= 0) || ((size_t
)sizeof(char) > (9223372036854775807L / (size_t)((3))))) ?
0 : (sizeof(char) * ((3))))))
;
1595 buf[0] = '\\';
1596 buf[1] = r;
1597 buf[2] = '\0';
1598 return buf;
1599 }
1600 buf = wmem_alloc_array(allocator, char, 5)((char*)wmem_alloc((allocator), (((((5)) <= 0) || ((size_t
)sizeof(char) > (9223372036854775807L / (size_t)((5))))) ?
0 : (sizeof(char) * ((5))))))
;
1601 buf[0] = '\\';
1602 buf[1] = 'x';
1603 buf[2] = hex[((uint8_t)c >> 4) & 0xF];
1604 buf[3] = hex[((uint8_t)c >> 0) & 0xF];
1605 buf[4] = '\0';
1606 return buf;
1607}
1608
1609char*
1610ws_utf8_truncate(char *string, size_t len)
1611{
1612 char* last_char;
1613
1614 /* Ensure that it is null terminated */
1615 string[len] = '\0';
1616 last_char = g_utf8_find_prev_char(string, string + len);
1617 if (last_char != NULL((void*)0) && g_utf8_get_char_validated(last_char, -1) == (gunichar)-2) {
1618 /* The last UTF-8 character was truncated into a partial sequence. */
1619 *last_char = '\0';
1620 }
1621 return string;
1622}
1623
1624/* ASCII/EBCDIC conversion tables from
1625 * https://web.archive.org/web/20060813174742/http://www.room42.com/store/computer_center/code_tables.shtml
1626 */
1627#if 0
1628static const uint8_t ASCII_translate_EBCDIC [ 256 ] = {
1629 0x00, 0x01, 0x02, 0x03, 0x04, 0x05, 0x06, 0x07, 0x08,
1630 0x09, 0x0A, 0x0B, 0x0C, 0x0D, 0x0E, 0x0F,
1631 0x10, 0x11, 0x12, 0x13, 0x14, 0x15, 0x16, 0x17, 0x18,
1632 0x19, 0x1A, 0x1B, 0x1C, 0x1D, 0x1E, 0x1F,
1633 0x40, 0x5A, 0x7F, 0x7B, 0x5B, 0x6C, 0x50, 0x7D, 0x4D,
1634 0x5D, 0x5C, 0x4E, 0x6B, 0x60, 0x4B, 0x61,
1635 0xF0, 0xF1, 0xF2, 0xF3, 0xF4, 0xF5, 0xF6, 0xF7, 0xF8,
1636 0xF9, 0x7A, 0x5E, 0x4C, 0x7E, 0x6E, 0x6F,
1637 0x7C, 0xC1, 0xC2, 0xC3, 0xC4, 0xC5, 0xC6, 0xC7, 0xC8,
1638 0xC9, 0xD1, 0xD2, 0xD3, 0xD4, 0xD5, 0xD6,
1639 0xD7, 0xD8, 0xD9, 0xE2, 0xE3, 0xE4, 0xE5, 0xE6, 0xE7,
1640 0xE8, 0xE9, 0xAD, 0xE0, 0xBD, 0x5F, 0x6D,
1641 0x7D, 0x81, 0x82, 0x83, 0x84, 0x85, 0x86, 0x87, 0x88,
1642 0x89, 0x91, 0x92, 0x93, 0x94, 0x95, 0x96,
1643 0x97, 0x98, 0x99, 0xA2, 0xA3, 0xA4, 0xA5, 0xA6, 0xA7,
1644 0xA8, 0xA9, 0xC0, 0x6A, 0xD0, 0xA1, 0x4B,
1645 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B,
1646 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B,
1647 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B,
1648 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B,
1649 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B,
1650 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B,
1651 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B,
1652 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B,
1653 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B,
1654 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B,
1655 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B,
1656 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B,
1657 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B,
1658 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B,
1659 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B,
1660 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B, 0x4B
1661};
1662
1663void
1664ASCII_to_EBCDIC(uint8_t *buf, unsigned bytes)
1665{
1666 unsigned i;
1667 uint8_t *bufptr;
1668
1669 bufptr = buf;
1670
1671 for (i = 0; i < bytes; i++, bufptr++) {
1672 *bufptr = ASCII_translate_EBCDIC[*bufptr];
1673 }
1674}
1675
1676uint8_t
1677ASCII_to_EBCDIC1(uint8_t c)
1678{
1679 return ASCII_translate_EBCDIC[c];
1680}
1681#endif
1682
1683static const uint8_t EBCDIC_translate_ASCII [ 256 ] = {
1684 0x00, 0x01, 0x02, 0x03, 0x04, 0x05, 0x06, 0x07,
1685 0x08, 0x09, 0x0A, 0x0B, 0x0C, 0x0D, 0x0E, 0x0F,
1686 0x10, 0x11, 0x12, 0x13, 0x14, 0x15, 0x16, 0x17,
1687 0x18, 0x19, 0x1A, 0x1B, 0x1C, 0x1D, 0x1E, 0x1F,
1688 0x20, 0x21, 0x22, 0x23, 0x24, 0x25, 0x26, 0x27,
1689 0x28, 0x29, 0x2A, 0x2B, 0x2C, 0x2D, 0x2E, 0x2F,
1690 0x2E, 0x2E, 0x32, 0x33, 0x34, 0x35, 0x36, 0x37,
1691 0x38, 0x39, 0x3A, 0x3B, 0x3C, 0x3D, 0x2E, 0x3F,
1692 0x20, 0x2E, 0x2E, 0x2E, 0x2E, 0x2E, 0x2E, 0x2E,
1693 0x2E, 0x2E, 0x2E, 0x2E, 0x3C, 0x28, 0x2B, 0x7C,
1694 0x26, 0x2E, 0x2E, 0x2E, 0x2E, 0x2E, 0x2E, 0x2E,
1695 0x2E, 0x2E, 0x21, 0x24, 0x2A, 0x29, 0x3B, 0x5E,
1696 0x2D, 0x2F, 0x2E, 0x2E, 0x2E, 0x2E, 0x2E, 0x2E,
1697 0x2E, 0x2E, 0x7C, 0x2C, 0x25, 0x5F, 0x3E, 0x3F,
1698 0x2E, 0x2E, 0x2E, 0x2E, 0x2E, 0x2E, 0x2E, 0x2E,
1699 0x2E, 0x2E, 0x3A, 0x23, 0x40, 0x27, 0x3D, 0x22,
1700 0x2E, 0x61, 0x62, 0x63, 0x64, 0x65, 0x66, 0x67,
1701 0x68, 0x69, 0x2E, 0x2E, 0x2E, 0x2E, 0x2E, 0x2E,
1702 0x2E, 0x6A, 0x6B, 0x6C, 0x6D, 0x6E, 0x6F, 0x70,
1703 0x71, 0x72, 0x2E, 0x2E, 0x2E, 0x2E, 0x2E, 0x2E,
1704 0x2E, 0x7E, 0x73, 0x74, 0x75, 0x76, 0x77, 0x78,
1705 0x79, 0x7A, 0x2E, 0x2E, 0x2E, 0x5B, 0x2E, 0x2E,
1706 0x2E, 0x2E, 0x2E, 0x2E, 0x2E, 0x2E, 0x2E, 0x2E,
1707 0x2E, 0x2E, 0x2E, 0x2E, 0x2E, 0x5D, 0x2E, 0x2E,
1708 0x7B, 0x41, 0x42, 0x43, 0x44, 0x45, 0x46, 0x47,
1709 0x48, 0x49, 0x2E, 0x2E, 0x2E, 0x2E, 0x2E, 0x2E,
1710 0x7D, 0x4A, 0x4B, 0x4C, 0x4D, 0x4E, 0x4F, 0x50,
1711 0x51, 0x52, 0x2E, 0x2E, 0x2E, 0x2E, 0x2E, 0x2E,
1712 0x5C, 0x2E, 0x53, 0x54, 0x55, 0x56, 0x57, 0x58,
1713 0x59, 0x5A, 0x2E, 0x2E, 0x2E, 0x2E, 0x2E, 0x2E,
1714 0x30, 0x31, 0x32, 0x33, 0x34, 0x35, 0x36, 0x37,
1715 0x38, 0x39, 0x2E, 0x2E, 0x2E, 0x2E, 0x2E, 0x2E
1716};
1717
1718void
1719EBCDIC_to_ASCII(uint8_t *buf, unsigned bytes)
1720{
1721 unsigned i;
1722 uint8_t *bufptr;
1723
1724 bufptr = buf;
1725
1726 for (i = 0; i < bytes; i++, bufptr++) {
1727 *bufptr = EBCDIC_translate_ASCII[*bufptr];
1728 }
1729}
1730
1731uint8_t
1732EBCDIC_to_ASCII1(uint8_t c)
1733{
1734 return EBCDIC_translate_ASCII[c];
1735}
1736
1737/*
1738 * This routine is based on a routine created by Dan Lasley
1739 * <DLASLEY@PROMUS.com>.
1740 *
1741 * It was modified for Wireshark by Gilbert Ramirez and others.
1742 */
1743
1744#define MAX_OFFSET_LEN8 8 /* max length of hex offset of bytes */
1745#define BYTES_PER_LINE16 16 /* max byte values printed on a line */
1746#define HEX_DUMP_LEN(16*3) (BYTES_PER_LINE16*3)
1747 /* max number of characters hex dump takes -
1748 2 digits plus trailing blank */
1749#define DATA_DUMP_LEN((16*3) + 2 + 2 + 16) (HEX_DUMP_LEN(16*3) + 2 + 2 + BYTES_PER_LINE16)
1750 /* number of characters those bytes take;
1751 3 characters per byte of hex dump,
1752 2 blanks separating hex from ASCII,
1753 2 optional ASCII dump delimiters,
1754 1 character per byte of ASCII dump */
1755#define MAX_LINE_LEN(8 + 2 + ((16*3) + 2 + 2 + 16)) (MAX_OFFSET_LEN8 + 2 + DATA_DUMP_LEN((16*3) + 2 + 2 + 16))
1756 /* number of characters per line;
1757 offset, 2 blanks separating offset
1758 from data dump, data dump */
1759
1760bool_Bool
1761hex_dump_buffer(bool_Bool (*print_line)(void *, const char *), void *fp,
1762 const unsigned char *cp, unsigned length,
1763 hex_dump_enc encoding,
1764 unsigned ascii_option)
1765{
1766 register unsigned int ad, i, j, k, l;
1767 unsigned char c;
1768 char line[MAX_LINE_LEN(8 + 2 + ((16*3) + 2 + 2 + 16)) + 1];
1769 unsigned int use_digits;
1770
1771 static const char binhex[16] = {
1772 '0', '1', '2', '3', '4', '5', '6', '7',
1773 '8', '9', 'a', 'b', 'c', 'd', 'e', 'f'};
1774
1775 /*
1776 * How many of the leading digits of the offset will we supply?
1777 * We always supply at least 4 digits, but if the maximum offset
1778 * won't fit in 4 digits, we use as many digits as will be needed.
1779 */
1780 if (((length - 1) & 0xF0000000) != 0)
1781 use_digits = 8; /* need all 8 digits */
1782 else if (((length - 1) & 0x0F000000) != 0)
1783 use_digits = 7; /* need 7 digits */
1784 else if (((length - 1) & 0x00F00000) != 0)
1785 use_digits = 6; /* need 6 digits */
1786 else if (((length - 1) & 0x000F0000) != 0)
1787 use_digits = 5; /* need 5 digits */
1788 else
1789 use_digits = 4; /* we'll supply 4 digits */
1790
1791 ad = 0;
1792 i = 0;
1793 j = 0;
1794 k = 0;
1795 while (i < length) {
1796 if ((i & 15) == 0) {
1797 /*
1798 * Start of a new line.
1799 */
1800 j = 0;
1801 l = use_digits;
1802 do {
1803 l--;
1804 c = (ad >> (l*4)) & 0xF;
1805 line[j++] = binhex[c];
1806 } while (l != 0);
1807 line[j++] = ' ';
1808 line[j++] = ' ';
1809 memset(line+j, ' ', DATA_DUMP_LEN((16*3) + 2 + 2 + 16));
1810
1811 /*
1812 * Offset in line of ASCII dump.
1813 */
1814 k = j + HEX_DUMP_LEN(16*3) + 2;
1815 if (ascii_option == HEXDUMP_ASCII_DELIMIT(0x0001U))
1816 line[k++] = '|';
1817 }
1818 c = *cp++;
1819 line[j++] = binhex[c>>4];
1820 line[j++] = binhex[c&0xf];
1821 j++;
1822 if (ascii_option != HEXDUMP_ASCII_EXCLUDE(0x0002U) ) {
1823 if (encoding == HEXDUMP_ENC_EBCDIC) {
1824 c = EBCDIC_to_ASCII1(c);
1825 }
1826 line[k++] = ((c >= ' ') && (c < 0x7f)) ? c : '.';
1827 }
1828 i++;
1829 if (((i & 15) == 0) || (i == length)) {
1830 /*
1831 * We'll be starting a new line, or
1832 * we're finished printing this buffer;
1833 * dump out the line we've constructed,
1834 * and advance the offset.
1835 */
1836 if (ascii_option == HEXDUMP_ASCII_DELIMIT(0x0001U))
1837 line[k++] = '|';
1838 line[k] = '\0';
1839 if (!print_line(fp, line))
1840 return false0;
1841 ad += 16;
1842 }
1843 }
1844 return true1;
1845}
1846
1847/*
1848 * Editor modelines - https://www.wireshark.org/tools/modelines.html
1849 *
1850 * Local variables:
1851 * c-basic-offset: 4
1852 * tab-width: 8
1853 * indent-tabs-mode: nil
1854 * End:
1855 *
1856 * vi: set shiftwidth=4 tabstop=8 expandtab:
1857 * :indentSize=4:tabSize=8:noTabs=true:
1858 */