| ... | ... | @@ -2,7 +2,8 @@ |
| 2 | 2 | //! |
| 3 | 3 | //! The source of an `error.WriteFailed` is always the backing writer. After an |
| 4 | 4 | //! `error.WriteFailed`, the `.writer` becomes `.failing` and is unrecoverable. |
| 5 | | //! After a `flush`, the writer also becomes `.failing` since the stream has |
| 5 | //! |
| 6 | //! After `finish`, the writer also becomes `.failing` since the stream has |
| 6 | 7 | //! been finished. This behavior also applies to `Raw` and `Huffman`. |
| 7 | 8 | |
| 8 | 9 | // Implementation details: |
| ... | ... | @@ -43,9 +44,10 @@ const PackedOptionalU15 = packed struct(u16) { |
| 43 | 44 | pub const null_bit: PackedOptionalU15 = .{ .value = 0, .is_null = true }; |
| 44 | 45 | }; |
| 45 | 46 | |
| 46 | | /// After `flush` is called, all vtable calls with result in `error.WriteFailed.` |
| 47 | /// After `finish` is called, all vtable calls with result in `error.WriteFailed`. |
| 47 | 48 | writer: Writer, |
| 48 | | has_history: bool, |
| 49 | history_len: u16, |
| 50 | history_end_unhashed: bool, |
| 49 | 51 | bit_writer: BitWriter, |
| 50 | 52 | buffered_tokens: struct { |
| 51 | 53 | /// List of `TokenBufferEntryHeader`s and their trailing data. |
| ... | ... | @@ -108,7 +110,7 @@ const BitWriter = struct { |
| 108 | 110 | b.buffered = @intCast(combined >> (combined_bits - b.buffered_n)); |
| 109 | 111 | } |
| 110 | 112 | |
| 111 | | /// Assserts one byte can be written to `b.otuput` without rebasing. |
| 113 | /// Asserts one byte can be written to `b.output` without rebasing. |
| 112 | 114 | pub fn byteAlign(b: *BitWriter) void { |
| 113 | 115 | b.output.unusedCapacitySlice()[0] = b.buffered; |
| 114 | 116 | b.output.advance(@intFromBool(b.buffered_n != 0)); |
| ... | ... | @@ -116,6 +118,35 @@ const BitWriter = struct { |
| 116 | 118 | b.buffered_n = 0; |
| 117 | 119 | } |
| 118 | 120 | |
| 121 | /// Byte align using only empty flate blocks |
| 122 | pub fn byteAlignBlocks(b: *BitWriter) Writer.Error!void { |
| 123 | if (b.buffered_n == 0) return; |
| 124 | |
| 125 | // There are two methods to do this: |
| 126 | // 1. A store block (5 or 6 bytes) |
| 127 | // 2. Outputting empty 10-bit fixed blocks until aligned |
| 128 | // |
| 129 | // Fixed blocks advance the bit alignment by two, and so can only used for even numbers |
| 130 | // requiring a maximum of four bytes (three blocks = 30 bits) to which is always more |
| 131 | // efficient than store blocks. |
| 132 | if (b.buffered_n & 1 == 0) { |
| 133 | const splat = (8 - @as(u5, b.buffered_n)) >> 1; |
| 134 | const bits = splat * 10; |
| 135 | // fixed eos code is 0, so the only bits are for the block header |
| 136 | const pattern: u32 = BlockHeader.int(.{ .kind = .fixed, .final = false }); |
| 137 | const splatted = ((pattern << 20) | (pattern << 10) | pattern) >> (30 - bits); |
| 138 | try b.write(splatted, bits); |
| 139 | } else { |
| 140 | try b.write(BlockHeader.int(.{ .kind = .stored, .final = false }), 3); |
| 141 | try b.output.rebase(0, 5); |
| 142 | b.byteAlign(); |
| 143 | b.output.writeInt(u16, 0x0000, .little) catch unreachable; |
| 144 | b.output.writeInt(u16, 0xffff, .little) catch unreachable; |
| 145 | } |
| 146 | |
| 147 | assert(b.buffered_n == 0); |
| 148 | } |
| 149 | |
| 119 | 150 | pub fn writeClen( |
| 120 | 151 | b: *BitWriter, |
| 121 | 152 | hclen: u4, |
| ... | ... | @@ -159,9 +190,9 @@ const BitWriter = struct { |
| 159 | 190 | /// The maximum value is `math.maxInt(u16) - 1` since one token is reserved for end-of-block. |
| 160 | 191 | const block_tokens: u16 = 1 << 15; |
| 161 | 192 | const lookup_hash_bits = 15; |
| 162 | | const Hash = u16; // `u[lookup_hash_bits]` is not used due to worse optimization (with LLVM 21) |
| 193 | const Hash = u16; // `@Int(.unsigned, lookup_hash_bits)` is not used due to worse optimization (with LLVM 21) |
| 163 | 194 | const seq_bytes = 3; // not intended to be changed |
| 164 | | const Seq = std.meta.Int(.unsigned, seq_bytes * 8); |
| 195 | const Seq = @Int(.unsigned, seq_bytes * 8); |
| 165 | 196 | |
| 166 | 197 | const TokenBufferEntryHeader = packed struct(u16) { |
| 167 | 198 | kind: enum(u1) { |
| ... | ... | @@ -295,7 +326,8 @@ pub fn init( |
| 295 | 326 | .rebase = rebase, |
| 296 | 327 | }, |
| 297 | 328 | }, |
| 298 | | .has_history = false, |
| 329 | .history_len = 0, |
| 330 | .history_end_unhashed = false, |
| 299 | 331 | .bit_writer = .init(output), |
| 300 | 332 | .buffered_tokens = .empty, |
| 301 | 333 | .lookup = .{ |
| ... | ... | @@ -314,68 +346,110 @@ fn drain(w: *Writer, data: []const []const u8, splat: usize) Writer.Error!usize |
| 314 | 346 | errdefer w.* = .failing; |
| 315 | 347 | // There may have not been enough space in the buffer and the write was sent directly here. |
| 316 | 348 | // However, it is required that all data goes through the buffer to keep a history. |
| 317 | | // |
| 318 | | // Additionally, ensuring the buffer is always full ensures there is always a full history |
| 319 | | // after. |
| 320 | 349 | const data_n = w.buffer.len - w.end; |
| 321 | 350 | _ = w.fixedDrain(data, splat) catch {}; |
| 322 | 351 | assert(w.end == w.buffer.len); |
| 323 | | try rebaseInner(w, 0, 1, false); |
| 352 | try rebaseInner(w, 0, 1, false, false); |
| 324 | 353 | return data_n; |
| 325 | 354 | } |
| 326 | 355 | |
| 327 | 356 | fn flush(w: *Writer) Writer.Error!void { |
| 328 | | defer w.* = .failing; |
| 357 | errdefer w.* = .failing; |
| 358 | try rebaseInner(w, 0, w.buffer.len - flate.history_len, true, false); |
| 329 | 359 | const c: *Compress = @fieldParentPtr("writer", w); |
| 330 | | try rebaseInner(w, 0, w.buffer.len - flate.history_len, true); |
| 360 | try c.bit_writer.byteAlignBlocks(); |
| 361 | } |
| 362 | |
| 363 | pub fn finish(c: *Compress) Writer.Error!void { |
| 364 | defer c.writer = .failing; |
| 365 | try rebaseInner(&c.writer, 0, c.writer.buffer.len - flate.history_len, true, true); |
| 331 | 366 | try c.bit_writer.output.rebase(0, 1); |
| 332 | 367 | c.bit_writer.byteAlign(); |
| 333 | 368 | try c.hasher.writeFooter(c.bit_writer.output); |
| 334 | 369 | } |
| 335 | 370 | |
| 336 | 371 | fn rebase(w: *Writer, preserve: usize, capacity: usize) Writer.Error!void { |
| 337 | | return rebaseInner(w, preserve, capacity, false); |
| 372 | errdefer w.* = .failing; |
| 373 | return rebaseInner(w, preserve, capacity, false, false); |
| 338 | 374 | } |
| 339 | 375 | |
| 340 | 376 | pub const rebase_min_preserve = flate.history_len; |
| 341 | 377 | pub const rebase_reserved_capacity = (token.max_length + 1) + seq_bytes; |
| 342 | 378 | |
| 343 | | fn rebaseInner(w: *Writer, preserve: usize, capacity: usize, eos: bool) Writer.Error!void { |
| 344 | | if (!eos) { |
| 379 | fn rebaseInner( |
| 380 | w: *Writer, |
| 381 | preserve: usize, |
| 382 | capacity: usize, |
| 383 | is_flush: bool, |
| 384 | is_finish: bool, |
| 385 | ) Writer.Error!void { |
| 386 | if (!is_flush) { |
| 345 | 387 | assert(@max(preserve, rebase_min_preserve) + (capacity + rebase_reserved_capacity) <= w.buffer.len); |
| 346 | | assert(w.end >= flate.history_len + rebase_reserved_capacity); // Above assert should |
| 347 | | // fail since rebase is only called when `capacity` is not present. This assertion is |
| 348 | | // important because a full history is required at the end. |
| 349 | 388 | } else { |
| 389 | // Preverse is not considered for `matching_end` |
| 350 | 390 | assert(preserve == 0 and capacity == w.buffer.len - flate.history_len); |
| 351 | 391 | } |
| 392 | if (is_finish) assert(is_flush); |
| 352 | 393 | |
| 353 | 394 | const c: *Compress = @fieldParentPtr("writer", w); |
| 354 | 395 | const buffered = w.buffered(); |
| 355 | 396 | |
| 356 | | const start = @as(usize, flate.history_len) * @intFromBool(c.has_history); |
| 357 | | const lit_end: usize = if (!eos) |
| 397 | const start: usize = c.history_len; |
| 398 | const hashable_len = buffered.len -| (seq_bytes - 1); |
| 399 | const matching_end: usize = if (!is_flush) |
| 358 | 400 | buffered.len - rebase_reserved_capacity - (preserve -| flate.history_len) |
| 359 | 401 | else |
| 360 | | buffered.len -| (seq_bytes - 1); |
| 402 | hashable_len; |
| 361 | 403 | |
| 362 | 404 | var i = start; |
| 363 | 405 | var last_unmatched = i; |
| 364 | | // Read from `w.buffer` instead of `buffered` since the latter may not |
| 365 | | // have enough bytes. If this is the case, this variable is not used. |
| 366 | | var seq: Seq = mem.readInt( |
| 367 | | std.meta.Int(.unsigned, (seq_bytes - 1) * 8), |
| 368 | | w.buffer[i..][0 .. seq_bytes - 1], |
| 369 | | .big, |
| 370 | | ); |
| 371 | | if (buffered[i..].len < seq_bytes - 1) { |
| 372 | | @branchHint(.unlikely); |
| 373 | | assert(eos); |
| 374 | | seq = undefined; |
| 375 | | assert(i >= lit_end); |
| 376 | | } |
| 406 | var seq: Seq = start_seq: { |
| 407 | if (c.history_end_unhashed) { |
| 408 | @branchHint(.unlikely); |
| 409 | |
| 410 | assert(i != 0); |
| 411 | i -|= seq_bytes - 1; |
| 412 | var seq: Seq = mem.readInt( |
| 413 | @Int(.unsigned, (seq_bytes - 1) * 8), |
| 414 | w.buffer[i..][0 .. seq_bytes - 1], |
| 415 | .big, |
| 416 | ); |
| 417 | |
| 418 | while (i < @min(start, hashable_len)) { |
| 419 | seq <<= 8; |
| 420 | seq |= buffered[i + (seq_bytes - 1)]; |
| 421 | c.addHash(i, hash(seq)); |
| 422 | i += 1; |
| 423 | } |
| 424 | |
| 425 | if (i < start) { |
| 426 | @branchHint(.unlikely); |
| 427 | i = start; |
| 428 | assert(i >= hashable_len); |
| 429 | assert(i >= matching_end); |
| 430 | assert(is_flush); |
| 431 | break :start_seq undefined; // Unused |
| 432 | } |
| 433 | |
| 434 | c.history_end_unhashed = false; |
| 435 | break :start_seq seq; |
| 436 | } |
| 437 | |
| 438 | if (i >= hashable_len) { |
| 439 | @branchHint(.unlikely); |
| 440 | assert(i >= matching_end); |
| 441 | assert(is_flush); |
| 442 | break :start_seq undefined; // Unused |
| 443 | } |
| 377 | 444 | |
| 378 | | while (i < lit_end) { |
| 445 | break :start_seq mem.readInt( |
| 446 | @Int(.unsigned, (seq_bytes - 1) * 8), |
| 447 | buffered[i..][0 .. seq_bytes - 1], |
| 448 | .big, |
| 449 | ); |
| 450 | }; |
| 451 | |
| 452 | while (i < matching_end) { |
| 379 | 453 | var match_start = i; |
| 380 | 454 | seq <<= 8; |
| 381 | 455 | seq |= buffered[i + (seq_bytes - 1)]; |
| ... | ... | @@ -420,43 +494,50 @@ fn rebaseInner(w: *Writer, preserve: usize, capacity: usize, eos: bool) Writer.E |
| 420 | 494 | |
| 421 | 495 | try c.outputBytes(buffered[last_unmatched..match_start]); |
| 422 | 496 | try c.outputMatch(@intCast(match.dist), @intCast(match.len - 3)); |
| 423 | | |
| 424 | 497 | last_unmatched = match_start + match.len; |
| 425 | | if (last_unmatched + seq_bytes >= w.end) { |
| 426 | | @branchHint(.unlikely); |
| 427 | | assert(eos); |
| 428 | | i = undefined; |
| 429 | | break; |
| 430 | | } |
| 431 | 498 | |
| 432 | | while (true) { |
| 499 | while (i < hashable_len) { |
| 433 | 500 | seq <<= 8; |
| 434 | 501 | seq |= buffered[i + (seq_bytes - 1)]; |
| 435 | | _ = c.addHash(i, hash(seq)); |
| 502 | c.addHash(i, hash(seq)); |
| 436 | 503 | i += 1; |
| 437 | 504 | |
| 438 | 505 | match_unadded -= 1; |
| 439 | 506 | if (match_unadded == 0) break; |
| 507 | } else { |
| 508 | @branchHint(.unlikely); |
| 509 | assert(is_flush); |
| 510 | // `c.history_end_unhashed` is set down below |
| 511 | break; |
| 440 | 512 | } |
| 441 | 513 | assert(i == match_start + match.len); |
| 442 | 514 | } |
| 443 | 515 | |
| 444 | | if (eos) { |
| 445 | | i = undefined; // (from match hashing logic) |
| 516 | if (is_flush) { |
| 446 | 517 | try c.outputBytes(buffered[last_unmatched..]); |
| 447 | 518 | c.hasher.update(buffered[start..]); |
| 448 | | try c.writeBlock(true); |
| 449 | | return; |
| 450 | | } |
| 451 | 519 | |
| 452 | | try c.outputBytes(buffered[last_unmatched..i]); |
| 453 | | c.hasher.update(buffered[start..i]); |
| 520 | if (is_finish) { |
| 521 | try c.writeBlock(true); |
| 522 | return; // Other state does not need updated since the writer transitions to `.failing` |
| 523 | } |
| 524 | |
| 525 | i = buffered.len; |
| 526 | c.history_end_unhashed = i != 0; |
| 527 | |
| 528 | if (c.buffered_tokens.n != 0) { |
| 529 | try c.writeBlock(false); |
| 530 | } |
| 531 | } else { |
| 532 | try c.outputBytes(buffered[last_unmatched..i]); |
| 533 | c.hasher.update(buffered[start..i]); |
| 534 | } |
| 454 | 535 | |
| 455 | | const preserved = buffered[i - flate.history_len ..]; |
| 456 | | assert(preserved.len > @max(rebase_min_preserve, preserve)); |
| 536 | c.history_len = @min(i, flate.history_len); |
| 537 | const preserved = buffered[i - c.history_len ..]; |
| 538 | if (!is_flush) assert(preserved.len >= @max(rebase_min_preserve, preserve)); |
| 457 | 539 | @memmove(w.buffer[0..preserved.len], preserved); |
| 458 | 540 | w.end = preserved.len; |
| 459 | | c.has_history = true; |
| 460 | 541 | } |
| 461 | 542 | |
| 462 | 543 | fn addHash(c: *Compress, i: usize, h: Hash) void { |
| ... | ... | @@ -499,7 +580,7 @@ fn betterMatchLen(old: u16, prev: []const u8, bytes: []const u8) u16 { |
| 499 | 580 | assert(bytes.len >= token.min_length); |
| 500 | 581 | |
| 501 | 582 | var i: u16 = 0; |
| 502 | | const Block = std.meta.Int(.unsigned, @min(math.divCeil( |
| 583 | const Block = @Int(.unsigned, @min(math.divCeil( |
| 503 | 584 | comptime_int, |
| 504 | 585 | math.ceilPowerOfTwoAssert(usize, @bitSizeOf(usize)), |
| 505 | 586 | 8, |
| ... | ... | @@ -798,7 +879,6 @@ test buildClen { |
| 798 | 879 | |
| 799 | 880 | fn writeBlock(c: *Compress, eos: bool) Writer.Error!void { |
| 800 | 881 | const toks = &c.buffered_tokens; |
| 801 | | if (!eos) assert(toks.n == block_tokens); |
| 802 | 882 | assert(toks.lit_freqs[256] == 0); |
| 803 | 883 | toks.lit_freqs[256] = 1; |
| 804 | 884 | |
| ... | ... | @@ -1438,20 +1518,8 @@ fn testFuzzedCompressInput(fbufs: *const [2][65536]u8, smith: *std.testing.Smith |
| 1438 | 1518 | .chain = chain, |
| 1439 | 1519 | }); |
| 1440 | 1520 | |
| 1441 | | // It is ensured that more bytes are not written then this to ensure this run |
| 1442 | | // does not take too long and that `flate_buf` does not run out of space. |
| 1443 | | const flate_buf_blocks = flate_buf.len / block_tokens; |
| 1444 | | // Allow a max overhead of 64 bytes per block since the implementation does not gaurauntee it |
| 1445 | | // writes store blocks when optimal. This comes from taking less than 32 bytes to write an |
| 1446 | | // optimal dynamic block header of mostly bitlen 8 codes and the end of block literal plus |
| 1447 | | // `(65536 / 256) / 8`, which is is the maximum number of extra bytes from bitlen 9 codes. An |
| 1448 | | // extra 32 bytes is reserved on top of that for container headers and footers. |
| 1449 | | const max_size = flate_buf.len - (flate_buf_blocks * 64 + 32); |
| 1450 | | |
| 1521 | var max_output: usize = 32; // Headers / footer |
| 1451 | 1522 | while (!smith.eosWeightedSimple(7, 1)) { |
| 1452 | | const max_bytes = max_size -| expected_size; |
| 1453 | | if (max_bytes == 0) break; |
| 1454 | | |
| 1455 | 1523 | const buffered = deflate_w.writer.buffered(); |
| 1456 | 1524 | // Required for repeating patterns and since writing from `buffered` is illegal |
| 1457 | 1525 | var copy_buf: [512]u8 = undefined; |
| ... | ... | @@ -1459,13 +1527,13 @@ fn testFuzzedCompressInput(fbufs: *const [2][65536]u8, smith: *std.testing.Smith |
| 1459 | 1527 | const bytes = bytes: switch (smith.valueRangeAtMost( |
| 1460 | 1528 | u2, |
| 1461 | 1529 | @intFromBool(buffered.len == 0), |
| 1462 | | 2, |
| 1530 | 3, |
| 1463 | 1531 | )) { |
| 1464 | 1532 | 0 => { // Copy |
| 1465 | 1533 | const start = smith.valueRangeLessThan(u32, 0, @intCast(buffered.len)); |
| 1466 | 1534 | // Reuse the implementation's history; otherwise, our own would need maintained. |
| 1467 | 1535 | const from = buffered[start..]; |
| 1468 | | const len = smith.valueRangeAtMost(u16, 1, @min(copy_buf.len, max_bytes)); |
| 1536 | const len = smith.valueRangeAtMost(u16, 1, copy_buf.len); |
| 1469 | 1537 | |
| 1470 | 1538 | const history_bytes = from[0..@min(from.len, len)]; |
| 1471 | 1539 | @memcpy(copy_buf[0..history_bytes.len], history_bytes); |
| ... | ... | @@ -1485,7 +1553,7 @@ fn testFuzzedCompressInput(fbufs: *const [2][65536]u8, smith: *std.testing.Smith |
| 1485 | 1553 | .value(FreqBufIndex, .random, 1), |
| 1486 | 1554 | }) |
| 1487 | 1555 | ]; |
| 1488 | | const len = smith.valueRangeAtMost(u32, 1, @min(fbuf.len, max_bytes)); |
| 1556 | const len = smith.valueRangeAtMost(u32, 1, fbuf.len); |
| 1489 | 1557 | const off = smith.valueRangeAtMost(u32, 0, @intCast(fbuf.len - len)); |
| 1490 | 1558 | break :bytes fbuf[off..][0..len]; |
| 1491 | 1559 | }, |
| ... | ... | @@ -1493,25 +1561,42 @@ fn testFuzzedCompressInput(fbufs: *const [2][65536]u8, smith: *std.testing.Smith |
| 1493 | 1561 | const rebaseable = bufsize - rebase_reserved_capacity; |
| 1494 | 1562 | const capacity = smith.valueRangeAtMost(u32, 1, rebaseable - rebase_min_preserve); |
| 1495 | 1563 | const preserve = smith.valueRangeAtMost(u32, 0, rebaseable - capacity); |
| 1496 | | try deflate_w.writer.rebase(preserve, capacity); |
| 1564 | const failed = deflate_w.writer.rebase(preserve, capacity); |
| 1565 | if (flate_w.buffered().len > max_output) return error.OverheadTooLarge; |
| 1566 | failed catch return; // Wrote too much data and ran out of space |
| 1567 | continue; |
| 1568 | }, |
| 1569 | 3 => { // Flush |
| 1570 | max_output += 8; // Alignment data |
| 1571 | const failed = deflate_w.writer.flush(); |
| 1572 | if (flate_w.buffered().len > max_output) return error.OverheadTooLarge; |
| 1573 | failed catch return; // Wrote too much data and ran out of space |
| 1497 | 1574 | continue; |
| 1498 | 1575 | }, |
| 1499 | | else => unreachable, |
| 1500 | 1576 | }; |
| 1501 | 1577 | |
| 1502 | | assert(bytes.len <= max_bytes); |
| 1503 | | try deflate_w.writer.writeAll(bytes); |
| 1578 | // An overhead of 64 bytes is given for each block since the implementation does not |
| 1579 | // gaurauntee it writes store blocks when optimal. This comes from taking less than 32 |
| 1580 | // bytes to write an optimal dynamic block header of mostly bitlen 8 codes and the end |
| 1581 | // of block literal plus `(65536 / 256) / 8`, which is is the maximum number of extra |
| 1582 | // bytes from bitlen 9 codes. |
| 1583 | max_output += bytes.len + ((bytes.len + flate_buf.len - 1) / block_tokens) * 64; |
| 1584 | const failed = deflate_w.writer.writeAll(bytes); |
| 1585 | if (flate_w.buffered().len > max_output) return error.OverheadTooLarge; |
| 1586 | failed catch return; // Wrote too much data and ran out of space |
| 1504 | 1587 | expected_hash.update(bytes); |
| 1505 | 1588 | expected_size += @intCast(bytes.len); |
| 1506 | 1589 | } |
| 1507 | 1590 | |
| 1508 | | try deflate_w.writer.flush(); |
| 1591 | const failed = deflate_w.finish(); |
| 1592 | if (flate_w.buffered().len > max_output) return error.OverheadTooLarge; |
| 1593 | failed catch return; // Wrote too much data and ran out of space |
| 1509 | 1594 | try testingCheckDecompressedMatches(flate_w.buffered(), expected_size, expected_hash); |
| 1510 | 1595 | } |
| 1511 | 1596 | |
| 1512 | 1597 | /// Does not compress data |
| 1513 | 1598 | pub const Raw = struct { |
| 1514 | | /// After `flush` is called, all vtable calls with result in `error.WriteFailed.` |
| 1599 | /// After `finish` is called, all vtable calls with result in `error.WriteFailed`. |
| 1515 | 1600 | writer: Writer, |
| 1516 | 1601 | output: *Writer, |
| 1517 | 1602 | hasher: flate.Container.Hasher, |
| ... | ... | @@ -1654,8 +1739,12 @@ pub const Raw = struct { |
| 1654 | 1739 | } |
| 1655 | 1740 | |
| 1656 | 1741 | fn flush(w: *Writer) Writer.Error!void { |
| 1657 | | defer w.* = .failing; |
| 1658 | | try Raw.rebaseInner(w, 0, w.buffer.len, true); |
| 1742 | errdefer w.* = .failing; |
| 1743 | try Raw.rebaseInner(w, 0, w.buffer.len, false); |
| 1744 | } |
| 1745 | |
| 1746 | fn finish(r: *Raw) Writer.Error!void { |
| 1747 | try Raw.rebaseInner(&r.writer, 0, r.writer.buffer.len, true); |
| 1659 | 1748 | } |
| 1660 | 1749 | |
| 1661 | 1750 | fn rebase(w: *Writer, preserve: usize, capacity: usize) Writer.Error!void { |
| ... | ... | @@ -1899,19 +1988,21 @@ fn testFuzzedRawInput(data_buf: *const [4 * 65536]u8, smith: *std.testing.Smith) |
| 1899 | 1988 | const Op = packed struct { |
| 1900 | 1989 | drain: bool = false, |
| 1901 | 1990 | add_vec: bool = false, |
| 1902 | | rebase: bool = false, |
| 1991 | rebase: enum(u2) { none, rebase, flush } = .none, |
| 1903 | 1992 | |
| 1904 | 1993 | pub const drain_only: @This() = .{ .drain = true }; |
| 1905 | 1994 | pub const add_vec_only: @This() = .{ .add_vec = true }; |
| 1906 | 1995 | pub const add_vec_and_drain: @This() = .{ .add_vec = true, .drain = true }; |
| 1907 | | pub const drain_and_rebase: @This() = .{ .drain = true, .rebase = true }; |
| 1996 | pub const drain_and_rebase: @This() = .{ .drain = true, .rebase = .rebase }; |
| 1997 | pub const drain_and_flush: @This() = .{ .drain = true, .rebase = .flush }; |
| 1908 | 1998 | }; |
| 1909 | 1999 | |
| 1910 | 2000 | const is_eos = expected_size == max_size or smith.eosWeightedSimple(7, 1); |
| 1911 | 2001 | var op: Op = if (!is_eos) smith.valueWeighted(Op, &.{ |
| 1912 | | .value(Op, .add_vec_only, 6), |
| 2002 | .value(Op, .add_vec_only, 5), |
| 1913 | 2003 | .value(Op, .add_vec_and_drain, 1), |
| 1914 | 2004 | .value(Op, .drain_and_rebase, 1), |
| 2005 | .value(Op, .drain_and_flush, 1), |
| 1915 | 2006 | }) else .drain_only; |
| 1916 | 2007 | |
| 1917 | 2008 | if (op.add_vec) { |
| ... | ... | @@ -1965,16 +2056,20 @@ fn testFuzzedRawInput(data_buf: *const [4 * 65536]u8, smith: *std.testing.Smith) |
| 1965 | 2056 | vecs_n = 0; |
| 1966 | 2057 | } |
| 1967 | 2058 | |
| 1968 | | if (op.rebase) { |
| 1969 | | const capacity = smith.valueRangeAtMost(u32, 0, raw_buf_len); |
| 1970 | | const preserve = smith.valueRangeAtMost(u32, 0, raw_buf_len - capacity); |
| 1971 | | try raw.writer.rebase(preserve, capacity); |
| 2059 | switch (op.rebase) { |
| 2060 | .none => {}, |
| 2061 | .rebase => { |
| 2062 | const capacity = smith.valueRangeAtMost(u32, 0, raw_buf_len); |
| 2063 | const preserve = smith.valueRangeAtMost(u32, 0, raw_buf_len - capacity); |
| 2064 | try raw.writer.rebase(preserve, capacity); |
| 2065 | }, |
| 2066 | .flush => try raw.writer.flush(), |
| 1972 | 2067 | } |
| 1973 | 2068 | |
| 1974 | 2069 | if (is_eos) break; |
| 1975 | 2070 | } |
| 1976 | 2071 | |
| 1977 | | try raw.writer.flush(); |
| 2072 | try raw.finish(); |
| 1978 | 2073 | try output.writer.flush(); |
| 1979 | 2074 | |
| 1980 | 2075 | try std.testing.expectEqual(.end, output.state); |
| ... | ... | @@ -1997,6 +2092,7 @@ fn testFuzzedRawInput(data_buf: *const [4 * 65536]u8, smith: *std.testing.Smith) |
| 1997 | 2092 | |
| 1998 | 2093 | /// Only performs huffman compression on data, does no matching. |
| 1999 | 2094 | pub const Huffman = struct { |
| 2095 | /// After `finish` is called, all vtable calls with result in `error.WriteFailed`. |
| 2000 | 2096 | writer: Writer, |
| 2001 | 2097 | bit_writer: BitWriter, |
| 2002 | 2098 | hasher: flate.Container.Hasher, |
| ... | ... | @@ -2026,12 +2122,6 @@ pub const Huffman = struct { |
| 2026 | 2122 | } |
| 2027 | 2123 | |
| 2028 | 2124 | fn drain(w: *Writer, data: []const []const u8, splat: usize) Writer.Error!usize { |
| 2029 | | { |
| 2030 | | //std.debug.print("drain {} (buffered)", .{w.buffered().len}); |
| 2031 | | //for (data) |d| std.debug.print("\n\t+ {}", .{d.len}); |
| 2032 | | //std.debug.print(" x {}\n\n", .{splat}); |
| 2033 | | } |
| 2034 | | |
| 2035 | 2125 | const h: *Huffman = @fieldParentPtr("writer", w); |
| 2036 | 2126 | const min_block = @min(w.buffer.len, max_tokens); |
| 2037 | 2127 | const pattern = data[data.len - 1]; |
| ... | ... | @@ -2238,9 +2328,15 @@ pub const Huffman = struct { |
| 2238 | 2328 | } |
| 2239 | 2329 | |
| 2240 | 2330 | fn flush(w: *Writer) Writer.Error!void { |
| 2241 | | defer w.* = .failing; |
| 2331 | errdefer w.* = .failing; |
| 2242 | 2332 | const h: *Huffman = @fieldParentPtr("writer", w); |
| 2243 | | try Huffman.rebaseInner(w, 0, w.buffer.len, true); |
| 2333 | try Huffman.rebaseInner(w, 0, w.buffer.len, false); |
| 2334 | try h.bit_writer.byteAlignBlocks(); |
| 2335 | } |
| 2336 | |
| 2337 | fn finish(h: *Huffman) Writer.Error!void { |
| 2338 | defer h.writer = .failing; |
| 2339 | try Huffman.rebaseInner(&h.writer, 0, h.writer.buffer.len, true); |
| 2244 | 2340 | try h.bit_writer.output.rebase(0, 1); |
| 2245 | 2341 | h.bit_writer.byteAlign(); |
| 2246 | 2342 | try h.hasher.writeFooter(h.bit_writer.output); |
| ... | ... | @@ -2359,9 +2455,6 @@ pub const Huffman = struct { |
| 2359 | 2455 | break :n stored_align_bits + @as(u32, 32) + @as(u32, bytes) * 8; |
| 2360 | 2456 | }; |
| 2361 | 2457 | |
| 2362 | | //std.debug.print("@ {}{{{}}} ", .{ h.bit_writer.output.end, h.bit_writer.buffered_n }); |
| 2363 | | //std.debug.print("#{} -> s {} f {} d {}\n", .{ bytes, stored_bitsize, fixed_bitsize, dynamic_bitsize }); |
| 2364 | | |
| 2365 | 2458 | if (stored_bitsize <= @min(dynamic_bitsize, fixed_bitsize)) { |
| 2366 | 2459 | try h.bit_writer.write(BlockHeader.int(.{ .kind = .stored, .final = eos }), 3); |
| 2367 | 2460 | try h.bit_writer.output.rebase(0, 5); |
| ... | ... | @@ -2434,19 +2527,21 @@ fn testFuzzedHuffmanInput(fbufs: *const [2][65536]u8, smith: *std.testing.Smith) |
| 2434 | 2527 | const Op = packed struct { |
| 2435 | 2528 | drain: bool = false, |
| 2436 | 2529 | add_vec: bool = false, |
| 2437 | | rebase: bool = false, |
| 2530 | rebase: enum(u2) { none, rebase, flush } = .none, |
| 2438 | 2531 | |
| 2439 | 2532 | pub const drain_only: @This() = .{ .drain = true }; |
| 2440 | 2533 | pub const add_vec_only: @This() = .{ .add_vec = true }; |
| 2441 | 2534 | pub const add_vec_and_drain: @This() = .{ .add_vec = true, .drain = true }; |
| 2442 | | pub const drain_and_rebase: @This() = .{ .drain = true, .rebase = true }; |
| 2535 | pub const drain_and_rebase: @This() = .{ .drain = true, .rebase = .rebase }; |
| 2536 | pub const drain_and_flush: @This() = .{ .drain = true, .rebase = .flush }; |
| 2443 | 2537 | }; |
| 2444 | 2538 | |
| 2445 | 2539 | const is_eos = expected_size == max_size or smith.eosWeightedSimple(7, 1); |
| 2446 | 2540 | var op: Op = if (!is_eos) smith.valueWeighted(Op, &.{ |
| 2447 | | .value(Op, .add_vec_only, 6), |
| 2541 | .value(Op, .add_vec_only, 5), |
| 2448 | 2542 | .value(Op, .add_vec_and_drain, 1), |
| 2449 | 2543 | .value(Op, .drain_and_rebase, 1), |
| 2544 | .value(Op, .drain_and_flush, 1), |
| 2450 | 2545 | }) else .drain_only; |
| 2451 | 2546 | |
| 2452 | 2547 | if (op.add_vec) { |
| ... | ... | @@ -2517,7 +2612,7 @@ fn testFuzzedHuffmanInput(fbufs: *const [2][65536]u8, smith: *std.testing.Smith) |
| 2517 | 2612 | vecs_n = 0; |
| 2518 | 2613 | } |
| 2519 | 2614 | |
| 2520 | | if (op.rebase) { |
| 2615 | if (op.rebase != .none) { |
| 2521 | 2616 | const capacity = smith.valueRangeAtMost(u32, 0, h_buf_len); |
| 2522 | 2617 | const preserve = smith.valueRangeAtMost(u32, 0, h_buf_len - capacity); |
| 2523 | 2618 | |
| ... | ... | @@ -2525,9 +2620,14 @@ fn testFuzzedHuffmanInput(fbufs: *const [2][65536]u8, smith: *std.testing.Smith) |
| 2525 | 2620 | h.writer.buffered().len, |
| 2526 | 2621 | flate_w.buffered().len, |
| 2527 | 2622 | false, |
| 2528 | | ); |
| 2529 | | h.writer.rebase(preserve, capacity) catch |
| 2530 | | return if (max_space <= flate_w.buffer.len) error.OverheadTooLarge else {}; |
| 2623 | ) + @as(usize, 8) * @intFromBool(op.rebase == .flush); // Overhead from byte alignment |
| 2624 | switch (op.rebase) { |
| 2625 | .none => unreachable, |
| 2626 | .rebase => h.writer.rebase(preserve, capacity) catch |
| 2627 | return if (max_space <= flate_w.buffer.len) error.OverheadTooLarge else {}, |
| 2628 | .flush => h.writer.flush() catch |
| 2629 | return if (max_space <= flate_w.buffer.len) error.OverheadTooLarge else {}, |
| 2630 | } |
| 2531 | 2631 | if (flate_w.buffered().len > max_space) return error.OverheadTooLarge; |
| 2532 | 2632 | } |
| 2533 | 2633 | |
| ... | ... | @@ -2539,8 +2639,7 @@ fn testFuzzedHuffmanInput(fbufs: *const [2][65536]u8, smith: *std.testing.Smith) |
| 2539 | 2639 | flate_w.buffered().len, |
| 2540 | 2640 | true, |
| 2541 | 2641 | ); |
| 2542 | | h.writer.flush() catch |
| 2543 | | return if (max_space <= flate_w.buffer.len) error.OverheadTooLarge else {}; |
| 2642 | h.finish() catch return if (max_space <= flate_w.buffer.len) error.OverheadTooLarge else {}; |
| 2544 | 2643 | if (flate_w.buffered().len > max_space) return error.OverheadTooLarge; |
| 2545 | 2644 | |
| 2546 | 2645 | try testingCheckDecompressedMatches(flate_w.buffered(), expected_size, expected_hash); |