dte test coverage


Directory: ./
Coverage: low: ≥ 0% medium: ≥ 50.0% high: ≥ 85.0%
Coverage Exec / Excl / Total
Lines: 53.3% 160 / 2 / 302
Functions: 71.4% 20 / 0 / 28
Branches: 39.0% 46 / 18 / 136

src/convert.c
Line Branch Exec Source
1 #include <errno.h>
2 #include <inttypes.h>
3 #include <stdlib.h>
4 #include <string.h>
5 #include "convert.h"
6 #include "block.h"
7 #include "buildvar-iconv.h"
8 #include "encoding.h"
9 #include "util/arith.h"
10 #include "util/debug.h"
11 #include "util/list.h"
12 #include "util/log.h"
13 #include "util/str-util.h"
14 #include "util/utf8.h"
15 #include "util/xmalloc.h"
16 #include "util/xreadwrite.h"
17
18 typedef struct {
19 StringView text;
20 size_t ipos;
21 struct CharsetConverter *cconv;
22 } FileDecoder;
23
24 56 static void add_block(Buffer *buffer, Block *blk)
25 {
26 56 buffer->nl += blk->nl;
27 56 list_insert_before(&blk->node, &buffer->blocks);
28 56 }
29
30 7824 static Block *add_utf8_line(Buffer *buffer, Block *blk, StringView line)
31 {
32 7824 const size_t len = line.length;
33 7824 size_t size = len + 1;
34
2/2
✓ Branch 2 → 3 taken 7796 times.
✓ Branch 2 → 6 taken 28 times.
7824 if (blk) {
35 7796 size_t avail = blk->alloc - blk->size;
36
2/2
✓ Branch 3 → 4 taken 7768 times.
✓ Branch 3 → 5 taken 28 times.
7796 if (size <= avail) {
37 7768 goto copy;
38 }
39 28 add_block(buffer, blk);
40 }
41
42 56 size = MAX(size, 8192);
43 56 blk = block_new(size);
44
45 7824 copy:
46 7824 memcpy(blk->data + blk->size, line.data, len);
47 7824 blk->size += len;
48 7824 blk->data[blk->size++] = '\n';
49 7824 blk->nl++;
50 7824 return blk;
51 }
52
53 7858 static bool read_utf8_line(FileDecoder *dec, StringView *linep)
54 {
55
2/2
✓ Branch 2 → 3 taken 34 times.
✓ Branch 2 → 5 taken 7824 times.
7858 if (dec->ipos >= dec->text.length) {
56 34 BUG_ON(dec->ipos > dec->text.length);
57 return false;
58 }
59
60 7824 *linep = get_delim(dec->text, &dec->ipos, '\n');
61 7824 return true;
62 }
63
64 34 static bool file_decoder_read_utf8(Buffer *buffer, StringView text, size_t *longest_line)
65 {
66
1/2
✗ Branch 3 → 4 not taken.
✓ Branch 3 → 5 taken 34 times.
34 if (unlikely(!encoding_is_utf8(buffer->encoding))) {
67 ✗ errno = EINVAL;
68 ✗ return false;
69 }
70
71 34 FileDecoder dec = {.text = text};
72 34 StringView line;
73
2/2
✓ Branch 6 → 7 taken 6 times.
✓ Branch 6 → 8 taken 28 times.
34 if (!read_utf8_line(&dec, &line)) {
74 6 *longest_line = 0;
75 6 return true;
76 }
77
78
2/2
✓ Branch 9 → 10 taken 1 time.
✓ Branch 9 → 11 taken 27 times.
28 if (strview_remove_matching_suffix(&line, "\r")) {
79 1 buffer->crlf_newlines = true;
80 }
81
82 28 Block *blk = add_utf8_line(buffer, NULL, line);
83 28 size_t maxline = line.length;
84
85
2/2
✓ Branch 12 → 16 taken 1 time.
✓ Branch 12 → 20 taken 27 times.
28 if (unlikely(buffer->crlf_newlines)) {
86
2/2
✓ Branch 17 → 13 taken 270 times.
✓ Branch 17 → 22 taken 1 time.
271 while (read_utf8_line(&dec, &line)) {
87 270 strview_remove_matching_suffix(&line, "\r");
88 270 blk = add_utf8_line(buffer, blk, line);
89 270 maxline = MAX(maxline, line.length);
90 }
91 } else {
92
2/2
✓ Branch 21 → 18 taken 7526 times.
✓ Branch 21 → 22 taken 27 times.
7553 while (read_utf8_line(&dec, &line)) {
93 7526 blk = add_utf8_line(buffer, blk, line);
94 7526 maxline = MAX(maxline, line.length);
95 }
96 }
97
98
1/2
✓ Branch 22 → 23 taken 28 times.
✗ Branch 22 → 24 not taken.
28 if (blk) {
99 28 add_block(buffer, blk);
100 }
101
102 28 *longest_line = maxline;
103 28 return true;
104 }
105
106 1 static size_t unix_to_dos(FileEncoder *enc, StringView text, size_t nr_newlines)
107 {
108 1 BUG_ON(text.length && !strview_has_suffix(text, "\n")); // See sanity_check_blocks()
109 1 BUG_ON(nr_newlines > text.length);
110
111 1 const size_t new_len = text.length + nr_newlines;
112
1/2
✓ Branch 8 → 9 taken 1 time.
✗ Branch 8 → 12 not taken.
1 if (enc->nsize < new_len) {
113 1 enc->nsize = xmul(text.length, 2);
114 1 enc->nbuf = xrealloc(enc->nbuf, enc->nsize);
115 }
116
117 1 size_t seen_nl = 0;
118 1 size_t dest_pos = 0;
119
120
2/2
✓ Branch 19 → 13 taken 3 times.
✓ Branch 19 → 20 taken 1 time.
4 for (size_t src_pos = 0; src_pos < text.length; ) {
121 3 const char *src = text.data + src_pos;
122 3 char *dest = enc->nbuf + dest_pos;
123 3 char *end = memccpy(dest, src, '\n', text.length - src_pos);
124 3 BUG_ON(!end); // Loop condition prevents this
125
126 3 size_t line_len = (size_t)(end - dest);
127 3 src_pos += line_len;
128 3 BUG_ON(src_pos > text.length);
129
130 3 end[-1] = '\r';
131 3 end[0] = '\n';
132 3 dest_pos += line_len + 1;
133 3 seen_nl++;
134 }
135
136 1 BUG_ON(seen_nl != nr_newlines);
137 1 BUG_ON(dest_pos != new_len);
138 1 return dest_pos;
139 }
140
141 #if ICONV_DISABLE == 1 // iconv not available; use basic, UTF-8 implementation:
142
143 bool conversion_supported_by_iconv (
144 const char* UNUSED_ARG(from),
145 const char* UNUSED_ARG(to)
146 ) {
147 errno = EINVAL;
148 return false;
149 }
150
151 FileEncoder file_encoder(const char *encoding, bool crlf, int fd)
152 {
153 if (unlikely(!encoding_is_utf8(encoding))) {
154 BUG("unsupported conversion; should have been handled earlier");
155 }
156
157 return (FileEncoder) {
158 .crlf = crlf,
159 .fd = fd,
160 };
161 }
162
163 void file_encoder_free(FileEncoder *enc)
164 {
165 free(enc->nbuf);
166 }
167
168 ssize_t file_encoder_write (
169 FileEncoder *enc,
170 const char *buf,
171 size_t size,
172 size_t nr_newlines
173 ) {
174 if (unlikely(enc->crlf)) {
175 size = unix_to_dos(enc, string_view(buf, size), nr_newlines);
176 buf = enc->nbuf;
177 }
178 return xwrite_all(enc->fd, buf, size);
179 }
180
181 size_t file_encoder_get_nr_errors(const FileEncoder* UNUSED_ARG(enc))
182 {
183 return 0;
184 }
185
186 bool file_decoder_read(Buffer *buffer, StringView text, size_t *longest_line)
187 {
188 return file_decoder_read_utf8(buffer, text, longest_line);
189 }
190
191 #else // ICONV_DISABLE != 1; use full iconv implementation:
192
193 #include <iconv.h>
194
195 // UTF-8 encoding of U+00BF (inverted question mark; "¿")
196 #define REPLACEMENT "\xc2\xbf"
197
198 typedef struct CharsetConverter {
199 iconv_t cd;
200 char *obuf;
201 size_t osize;
202 size_t opos;
203 size_t consumed;
204 size_t errors;
205
206 // Temporary input buffer
207 char tbuf[16];
208 size_t tcount;
209
210 // REPLACEMENT character, in target encoding
211 char rbuf[4];
212 size_t rcount;
213
214 // Input character size in bytes, or zero for UTF-8
215 size_t char_size;
216 } CharsetConverter;
217
218 1 static CharsetConverter *create(iconv_t cd)
219 {
220 1 CharsetConverter *c = xcalloc1(sizeof(*c));
221 1 c->cd = cd;
222 1 c->osize = 8192;
223 1 c->obuf = xmalloc(c->osize);
224 1 return c;
225 }
226
227 2 static size_t iconv_wrapper (
228 iconv_t cd,
229 const char **restrict inbuf,
230 size_t *restrict inbytesleft,
231 char **restrict outbuf,
232 size_t *restrict outbytesleft
233 ) {
234 // POSIX defines the second parameter of iconv(3) as `char **restrict` but
235 // NetBSD and Illumos declare it as `const char **restrict`, so we cast to
236 // `void*` here to prevent `-Wincompatible-pointer-types` errors.
237 // https://pubs.opengroup.org/onlinepubs/9799919799/functions/iconv.html#:~:text=char%20**restrict%20inbuf
238 2 return iconv(cd, (void*)inbuf, inbytesleft, outbuf, outbytesleft);
239 }
240
241 ✗ static void resize_obuf(CharsetConverter *c)
242 {
243 ✗ c->osize = xmul(2, c->osize);
244 ✗ c->obuf = xrealloc(c->obuf, c->osize);
245 ✗ }
246
247 ✗ static void add_replacement(CharsetConverter *c)
248 {
249 ✗ if (c->osize - c->opos < 4) {
250 ✗ resize_obuf(c);
251 }
252
253 ✗ memcpy(c->obuf + c->opos, c->rbuf, c->rcount);
254 ✗ c->opos += c->rcount;
255 ✗ }
256
257 ✗ static size_t handle_invalid(CharsetConverter *c, const char *buf, size_t count)
258 {
259 ✗ LOG_DEBUG("%zu %zu", c->char_size, count);
260 ✗ add_replacement(c);
261 ✗ if (c->char_size == 0) {
262 // Converting from UTF-8
263 ✗ size_t idx = 0;
264 ✗ CodePoint u = u_get_char(buf, count, &idx);
265 ✗ LOG_DEBUG("U+%04" PRIX32, u);
266 ✗ return idx;
267 }
268 ✗ if (c->char_size > count) {
269 // wtf
270 ✗ return 1;
271 }
272 return c->char_size;
273 }
274
275 1 static int xiconv(CharsetConverter *c, const char **ib, size_t *ic)
276 {
277 1 while (1) {
278 1 char *ob = c->obuf + c->opos;
279 1 size_t oc = c->osize - c->opos;
280 1 size_t rc = iconv_wrapper(c->cd, ib, ic, &ob, &oc);
281 1 c->opos = ob - c->obuf;
282
1/2
✗ Branch 4 → 5 not taken.
✓ Branch 4 → 12 taken 1 time.
1 if (rc == (size_t)-1) {
283 ✗ switch (errno) {
284 ✗ case EILSEQ:
285 ✗ c->errors++;
286 // Reset
287 ✗ iconv(c->cd, NULL, NULL, NULL, NULL);
288 ✗ return errno;
289 case EINVAL:
290 return errno;
291 ✗ case E2BIG:
292 ✗ resize_obuf(c);
293 ✗ continue;
294 ✗ default:
295 − BUG("iconv: %s", strerror(errno));
296 }
297 } else {
298 1 c->errors += rc;
299 }
300 1 return 0;
301 }
302 }
303
304 ✗ static size_t convert_incomplete(CharsetConverter *c, const char *input, size_t len)
305 {
306 ✗ size_t ipos = 0;
307 ✗ while (c->tcount < sizeof(c->tbuf) && ipos < len) {
308 ✗ c->tbuf[c->tcount++] = input[ipos++];
309 ✗ const char *ib = c->tbuf;
310 ✗ size_t ic = c->tcount;
311 ✗ int rc = xiconv(c, &ib, &ic);
312 ✗ if (ic > 0) {
313 ✗ memmove(c->tbuf, ib, ic);
314 }
315 ✗ c->tcount = ic;
316 ✗ if (rc == EINVAL) {
317 // Incomplete character at end of input buffer; try again
318 // with more input data
319 ✗ continue;
320 }
321 ✗ if (rc == EILSEQ) {
322 // Invalid multibyte sequence
323 ✗ size_t skip = handle_invalid(c, c->tbuf, c->tcount);
324 ✗ c->tcount -= skip;
325 ✗ if (c->tcount > 0) {
326 ✗ LOG_DEBUG("tcount=%zu, skip=%zu", c->tcount, skip);
327 ✗ memmove(c->tbuf, c->tbuf + skip, c->tcount);
328 ✗ continue;
329 }
330 ✗ return ipos;
331 }
332 ✗ break;
333 }
334
335 ✗ LOG_DEBUG("%zu %zu", ipos, c->tcount);
336 ✗ return ipos;
337 }
338
339 1 static void cconv_process(CharsetConverter *c, const char *input, size_t len)
340 {
341
1/2
✗ Branch 2 → 3 not taken.
✓ Branch 2 → 4 taken 1 time.
1 if (c->consumed > 0) {
342 ✗ size_t fill = c->opos - c->consumed;
343 ✗ memmove(c->obuf, c->obuf + c->consumed, fill);
344 ✗ c->opos = fill;
345 ✗ c->consumed = 0;
346 }
347
348
1/2
✗ Branch 4 → 5 not taken.
✓ Branch 4 → 7 taken 1 time.
1 if (c->tcount > 0) {
349 ✗ size_t ipos = convert_incomplete(c, input, len);
350 ✗ input += ipos;
351 ✗ len -= ipos;
352 }
353
354 1 const char *ib = input;
355
2/2
✓ Branch 17 → 8 taken 1 time.
✓ Branch 17 → 18 taken 1 time.
2 for (size_t ic = len; ic > 0; ) {
356 1 int r = xiconv(c, &ib, &ic);
357
1/2
✗ Branch 9 → 10 not taken.
✓ Branch 9 → 13 taken 1 time.
1 if (r == EINVAL) {
358 // Incomplete character at end of input buffer
359 ✗ if (ic < sizeof(c->tbuf)) {
360 ✗ memcpy(c->tbuf, ib, ic);
361 ✗ c->tcount = ic;
362 } else {
363 // FIXME
364 ✗ }
365 ✗ ic = 0;
366 ✗ continue;
367 }
368
1/2
✗ Branch 13 → 14 not taken.
✓ Branch 13 → 16 taken 1 time.
1 if (r == EILSEQ) {
369 // Invalid multibyte sequence
370 ✗ size_t skip = handle_invalid(c, ib, ic);
371 ✗ ic -= skip;
372 ✗ ib += skip;
373 ✗ continue;
374 }
375 }
376 1 }
377
378 ✗ static CharsetConverter *cconv_to_utf8(const char *encoding)
379 {
380 ✗ iconv_t cd = iconv_open("UTF-8", encoding);
381 ✗ if (cd == (iconv_t)-1) {
382 return NULL;
383 }
384
385 ✗ CharsetConverter *c = create(cd);
386 ✗ c->rcount = copyliteral(c->rbuf, REPLACEMENT);
387
388 ✗ if (str_has_prefix(encoding, "UTF-16")) {
389 ✗ c->char_size = 2;
390 ✗ } else if (str_has_prefix(encoding, "UTF-32")) {
391 ✗ c->char_size = 4;
392 } else {
393 ✗ c->char_size = 1;
394 }
395
396 return c;
397 }
398
399 1 static void encode_replacement(CharsetConverter *c)
400 {
401 1 static const char rep[] = REPLACEMENT;
402 1 const char *ib = rep;
403 1 char *ob = c->rbuf;
404 1 size_t ic = STRLEN(REPLACEMENT);
405 1 size_t oc = sizeof(c->rbuf);
406 1 size_t rc = iconv_wrapper(c->cd, &ib, &ic, &ob, &oc);
407
408
1/2
✓ Branch 3 → 4 taken 1 time.
✗ Branch 3 → 5 not taken.
1 if (rc == (size_t)-1) {
409 1 c->rbuf[0] = '\xbf';
410 1 c->rcount = 1;
411 } else {
412 ✗ c->rcount = ob - c->rbuf;
413 }
414 1 }
415
416 1 static CharsetConverter *cconv_from_utf8(const char *encoding)
417 {
418 1 iconv_t cd = iconv_open(encoding, "UTF-8");
419
1/2
✓ Branch 3 → 4 taken 1 time.
✗ Branch 3 → 7 not taken.
1 if (cd == (iconv_t)-1) {
420 return NULL;
421 }
422 1 CharsetConverter *c = create(cd);
423 1 encode_replacement(c);
424 1 return c;
425 }
426
427 1 static void cconv_flush(CharsetConverter *c)
428 {
429
1/2
✗ Branch 2 → 3 not taken.
✓ Branch 2 → 6 taken 1 time.
1 if (c->tcount > 0) {
430 // Replace incomplete character at end of input buffer
431 ✗ LOG_DEBUG("incomplete character at EOF");
432 ✗ add_replacement(c);
433 ✗ c->tcount = 0;
434 }
435 1 }
436
437 ✗ static char *cconv_consume_line(CharsetConverter *c, size_t *len)
438 {
439 ✗ char *line = c->obuf + c->consumed;
440 ✗ char *nl = memchr(line, '\n', c->opos - c->consumed);
441 ✗ if (!nl) {
442 ✗ *len = 0;
443 ✗ return NULL;
444 }
445
446 ✗ size_t n = nl - line + 1;
447 ✗ c->consumed += n;
448 ✗ *len = n;
449 ✗ return line;
450 }
451
452 1 static char *cconv_consume_all(CharsetConverter *c, size_t *len)
453 {
454 1 char *buf = c->obuf + c->consumed;
455 1 *len = c->opos - c->consumed;
456 1 c->consumed = c->opos;
457 1 return buf;
458 }
459
460 1 static void cconv_free(CharsetConverter *c)
461 {
462 1 BUG_ON(!c);
463 1 iconv_close(c->cd);
464 1 free(c->obuf);
465 1 free(c);
466 1 }
467
468 2 bool conversion_supported_by_iconv(const char *from, const char *to)
469 {
470
2/4
✓ Branch 2 → 3 taken 2 times.
✗ Branch 2 → 4 not taken.
✗ Branch 3 → 4 not taken.
✓ Branch 3 → 5 taken 2 times.
2 if (unlikely(from[0] == '\0' || to[0] == '\0')) {
471 ✗ errno = EINVAL;
472 ✗ return false;
473 }
474
475 2 iconv_t cd = iconv_open(to, from);
476
1/2
✓ Branch 6 → 7 taken 2 times.
✗ Branch 6 → 9 not taken.
2 if (cd == (iconv_t)-1) {
477 return false;
478 }
479
480 2 iconv_close(cd);
481 2 return true;
482 }
483
484 22 FileEncoder file_encoder(const char *encoding, bool crlf, int fd)
485 {
486 22 CharsetConverter *cconv = NULL;
487
2/2
✓ Branch 3 → 4 taken 1 time.
✓ Branch 3 → 7 taken 21 times.
22 if (unlikely(!encoding_is_utf8(encoding))) {
488 1 cconv = cconv_from_utf8(encoding);
489
1/2
✗ Branch 5 → 6 not taken.
✓ Branch 5 → 7 taken 1 time.
1 if (!cconv) {
490 − BUG("unsupported conversion; should have been handled earlier");
491 }
492 }
493
494 22 return (FileEncoder) {
495 .cconv = cconv,
496 .crlf = crlf,
497 .fd = fd,
498 };
499 }
500
501 22 void file_encoder_free(FileEncoder *enc)
502 {
503
2/2
✓ Branch 2 → 3 taken 1 time.
✓ Branch 2 → 4 taken 21 times.
22 if (enc->cconv) {
504 1 cconv_free(enc->cconv);
505 }
506 22 free(enc->nbuf);
507 22 }
508
509 // NOTE: buf must contain whole characters!
510 22 ssize_t file_encoder_write (
511 FileEncoder *enc,
512 const char *buf,
513 size_t size,
514 size_t nr_newlines
515 ) {
516
2/2
✓ Branch 2 → 3 taken 1 time.
✓ Branch 2 → 5 taken 21 times.
22 if (unlikely(enc->crlf)) {
517 1 size = unix_to_dos(enc, string_view(buf, size), nr_newlines);
518 1 buf = enc->nbuf;
519 }
520
2/2
✓ Branch 5 → 6 taken 1 time.
✓ Branch 5 → 9 taken 21 times.
22 if (unlikely(enc->cconv)) {
521 1 cconv_process(enc->cconv, buf, size);
522 1 cconv_flush(enc->cconv);
523 1 buf = cconv_consume_all(enc->cconv, &size);
524 }
525 22 return xwrite_all(enc->fd, buf, size);
526 }
527
528 22 size_t file_encoder_get_nr_errors(const FileEncoder *enc)
529 {
530
2/2
✓ Branch 2 → 3 taken 1 time.
✓ Branch 2 → 4 taken 21 times.
22 return enc->cconv ? enc->cconv->errors : 0;
531 }
532
533 ✗ static bool fill(FileDecoder *dec)
534 {
535 ✗ StringView text = dec->text;
536 ✗ if (dec->ipos == text.length) {
537 return false;
538 }
539
540 // Smaller than cconv.obuf to make realloc less likely
541 ✗ size_t max = 7 * 1024;
542
543 ✗ size_t icount = MIN(text.length - dec->ipos, max);
544 ✗ cconv_process(dec->cconv, text.data + dec->ipos, icount);
545 ✗ dec->ipos += icount;
546 ✗ if (dec->ipos == text.length) {
547 // Must be flushed after all input has been fed
548 ✗ cconv_flush(dec->cconv);
549 }
550 return true;
551 }
552
553 ✗ static bool decode_and_read_line(FileDecoder *dec, StringView *linep)
554 {
555 ✗ char *line;
556 ✗ size_t len;
557 ✗ while (1) {
558 ✗ line = cconv_consume_line(dec->cconv, &len);
559 ✗ if (line || !fill(dec)) {
560 break;
561 }
562 }
563
564 ✗ if (line) {
565 // Newline not wanted
566 ✗ len--;
567 } else {
568 ✗ line = cconv_consume_all(dec->cconv, &len);
569 ✗ if (len == 0) {
570 return false;
571 }
572 }
573
574 ✗ *linep = string_view(line, len);
575 ✗ return true;
576 }
577
578 34 bool file_decoder_read(Buffer *buffer, StringView text, size_t *longest_line)
579 {
580
1/2
✓ Branch 3 → 4 taken 34 times.
✗ Branch 3 → 5 not taken.
34 if (encoding_is_utf8(buffer->encoding)) {
581 34 return file_decoder_read_utf8(buffer, text, longest_line);
582 }
583
584 ✗ CharsetConverter *cconv = cconv_to_utf8(buffer->encoding);
585 ✗ if (!cconv) {
586 return false;
587 }
588
589 ✗ FileDecoder dec = {.text = text, .cconv = cconv};
590 ✗ StringView line;
591 ✗ if (!decode_and_read_line(&dec, &line)) {
592 ✗ *longest_line = 0;
593 ✗ cconv_free(cconv);
594 ✗ return true;
595 }
596
597 ✗ if (strview_remove_matching_suffix(&line, "\r")) {
598 ✗ buffer->crlf_newlines = true;
599 }
600
601 ✗ Block *blk = add_utf8_line(buffer, NULL, line);
602 ✗ size_t maxline = line.length;
603 ✗ while (decode_and_read_line(&dec, &line)) {
604 ✗ if (buffer->crlf_newlines) {
605 ✗ strview_remove_matching_suffix(&line, "\r");
606 }
607 ✗ blk = add_utf8_line(buffer, blk, line);
608 ✗ maxline = MAX(maxline, line.length);
609 }
610
611 ✗ if (blk) {
612 ✗ add_block(buffer, blk);
613 }
614
615 ✗ *longest_line = maxline;
616 ✗ cconv_free(cconv);
617 ✗ return true;
618 }
619
620 #endif
621