| ... | @@ -1,4 +1,3 @@ | ... | @@ -1,4 +1,3 @@ |
| 1 | const std = @import("std.zig"); | | |
| 2 | /// Tar archive is single ordinary file which can contain many files (or | 1 | /// Tar archive is single ordinary file which can contain many files (or |
| 3 | /// directories, symlinks, ...). It's build by series of blocks each size of 512 | 2 | /// directories, symlinks, ...). It's build by series of blocks each size of 512 |
| 4 | /// bytes. First block of each entry is header which defines type, name, size | 3 | /// bytes. First block of each entry is header which defines type, name, size |
| ... | @@ -15,7 +14,9 @@ const std = @import("std.zig"); | ... | @@ -15,7 +14,9 @@ const std = @import("std.zig"); |
| 15 | /// | 14 | /// |
| 16 | /// GNU tar reference: https://www.gnu.org/software/tar/manual/html_node/Standard.html | 15 | /// GNU tar reference: https://www.gnu.org/software/tar/manual/html_node/Standard.html |
| 17 | /// pax reference: https://pubs.opengroup.org/onlinepubs/9699919799/utilities/pax.html#tag_20_92_13 | 16 | /// pax reference: https://pubs.opengroup.org/onlinepubs/9699919799/utilities/pax.html#tag_20_92_13 |
| 18 | | 17 | /// |
| | 18 | //const std = @import("std.zig"); |
| | 19 | const std = @import("std"); |
| 19 | const assert = std.debug.assert; | 20 | const assert = std.debug.assert; |
| 20 | | 21 | |
| 21 | pub const Options = struct { | 22 | pub const Options = struct { |
| ... | @@ -224,338 +225,6 @@ inline fn blockPadding(size: usize) usize { | ... | @@ -224,338 +225,6 @@ inline fn blockPadding(size: usize) usize { |
| 224 | return block_rounded - size; | 225 | return block_rounded - size; |
| 225 | } | 226 | } |
| 226 | | 227 | |
| 227 | fn BufferedReader(comptime ReaderType: type) type { | | |
| 228 | return struct { | | |
| 229 | underlying_reader: ReaderType, | | |
| 230 | buffer: [BLOCK_SIZE * 8]u8 = undefined, | | |
| 231 | start: usize = 0, | | |
| 232 | end: usize = 0, | | |
| 233 | | | |
| 234 | const Self = @This(); | | |
| 235 | | | |
| 236 | // Fills buffer from underlying unbuffered reader. | | |
| 237 | fn fillBuffer(self: *Self) !void { | | |
| 238 | self.removeUsed(); | | |
| 239 | self.end += try self.underlying_reader.read(self.buffer[self.end..]); | | |
| 240 | } | | |
| 241 | | | |
| 242 | // Returns slice of size count or how much fits into buffer. | | |
| 243 | pub fn readSlice(self: *Self, count: usize) ![]const u8 { | | |
| 244 | if (count <= self.end - self.start) { | | |
| 245 | return self.buffer[self.start .. self.start + count]; | | |
| 246 | } | | |
| 247 | try self.fillBuffer(); | | |
| 248 | const buf = self.buffer[self.start..self.end]; | | |
| 249 | if (buf.len == 0) return error.UnexpectedEndOfStream; | | |
| 250 | return buf[0..@min(count, buf.len)]; | | |
| 251 | } | | |
| 252 | | | |
| 253 | // Returns tar header block, 512 bytes, or null if eof. Before reading | | |
| 254 | // advances buffer for padding of the previous block, to position reader | | |
| 255 | // at the start of new block. After reading advances for block size, to | | |
| 256 | // position reader at the start of the file content. | | |
| 257 | pub fn readHeader(self: *Self, padding: usize) !?[]const u8 { | | |
| 258 | try self.skip(padding); | | |
| 259 | const buf = self.readSlice(BLOCK_SIZE) catch return null; | | |
| 260 | if (buf.len < BLOCK_SIZE) return error.UnexpectedEndOfStream; | | |
| 261 | self.advance(BLOCK_SIZE); | | |
| 262 | return buf[0..BLOCK_SIZE]; | | |
| 263 | } | | |
| 264 | | | |
| 265 | // Returns byte at current position in buffer. | | |
| 266 | pub fn readByte(self: *@This()) u8 { | | |
| 267 | assert(self.start < self.end); | | |
| 268 | return self.buffer[self.start]; | | |
| 269 | } | | |
| 270 | | | |
| 271 | // Advances reader for count bytes, assumes that we have that number of | | |
| 272 | // bytes in buffer. | | |
| 273 | pub fn advance(self: *Self, count: usize) void { | | |
| 274 | self.start += count; | | |
| 275 | assert(self.start <= self.end); | | |
| 276 | } | | |
| 277 | | | |
| 278 | // Advances reader without assuming that count bytes are in the buffer. | | |
| 279 | pub fn skip(self: *Self, count: usize) !void { | | |
| 280 | if (self.start + count > self.end) { | | |
| 281 | try self.underlying_reader.skipBytes(self.start + count - self.end, .{}); | | |
| 282 | self.start = self.end; | | |
| 283 | } else { | | |
| 284 | self.advance(count); | | |
| 285 | } | | |
| 286 | } | | |
| 287 | | | |
| 288 | // Removes used part of the buffer. | | |
| 289 | inline fn removeUsed(self: *Self) void { | | |
| 290 | const dest_end = self.end - self.start; | | |
| 291 | if (self.start == 0 or dest_end > self.start) return; | | |
| 292 | @memcpy(self.buffer[0..dest_end], self.buffer[self.start..self.end]); | | |
| 293 | self.end = dest_end; | | |
| 294 | self.start = 0; | | |
| 295 | } | | |
| 296 | | | |
| 297 | // Writes count bytes to the writer. Advances reader. | | |
| 298 | pub fn write(self: *Self, writer: anytype, count: usize) !void { | | |
| 299 | var pos: usize = 0; | | |
| 300 | while (pos < count) { | | |
| 301 | const slice = try self.readSlice(count - pos); | | |
| 302 | try writer.writeAll(slice); | | |
| 303 | self.advance(slice.len); | | |
| 304 | pos += slice.len; | | |
| 305 | } | | |
| 306 | } | | |
| 307 | | | |
| 308 | // Copies dst.len bytes into dst buffer. Advances reader. | | |
| 309 | pub fn copy(self: *Self, dst: []u8) ![]const u8 { | | |
| 310 | var pos: usize = 0; | | |
| 311 | while (pos < dst.len) { | | |
| 312 | const slice = try self.readSlice(dst.len - pos); | | |
| 313 | @memcpy(dst[pos .. pos + slice.len], slice); | | |
| 314 | self.advance(slice.len); | | |
| 315 | pos += slice.len; | | |
| 316 | } | | |
| 317 | return dst; | | |
| 318 | } | | |
| 319 | | | |
| 320 | pub fn paxFileReader(self: *Self, size: usize) PaxFileReader { | | |
| 321 | return .{ | | |
| 322 | .size = size, | | |
| 323 | .reader = self, | | |
| 324 | .offset = 0, | | |
| 325 | }; | | |
| 326 | } | | |
| 327 | | | |
| 328 | const PaxFileReader = struct { | | |
| 329 | size: usize, | | |
| 330 | offset: usize = 0, | | |
| 331 | reader: *Self, | | |
| 332 | | | |
| 333 | const PaxKeyKind = enum { | | |
| 334 | path, | | |
| 335 | linkpath, | | |
| 336 | size, | | |
| 337 | }; | | |
| 338 | | | |
| 339 | const PaxAttribute = struct { | | |
| 340 | key: PaxKeyKind, | | |
| 341 | value_len: usize, | | |
| 342 | parent: *PaxFileReader, | | |
| 343 | | | |
| 344 | // Copies pax attribute value into destination buffer. | | |
| 345 | // Must be called with destination buffer of size at least value_len. | | |
| 346 | pub fn value(self: PaxAttribute, dst: []u8) ![]u8 { | | |
| 347 | assert(dst.len >= self.value_len); | | |
| 348 | const buf = dst[0..self.value_len]; | | |
| 349 | _ = try self.parent.reader.copy(buf); | | |
| 350 | self.parent.offset += buf.len; | | |
| 351 | try self.parent.checkAttributeEnding(); | | |
| 352 | return buf; | | |
| 353 | } | | |
| 354 | }; | | |
| 355 | | | |
| 356 | // Caller of the next has to call value in PaxAttribute, to advance | | |
| 357 | // reader across value. | | |
| 358 | pub fn next(self: *PaxFileReader) !?PaxAttribute { | | |
| 359 | while (true) { | | |
| 360 | const remaining_size = self.size - self.offset; | | |
| 361 | if (remaining_size == 0) return null; | | |
| 362 | | | |
| 363 | const inf = try parsePaxAttribute( | | |
| 364 | try self.reader.readSlice(remaining_size), | | |
| 365 | remaining_size, | | |
| 366 | ); | | |
| 367 | const key: PaxKeyKind = if (inf.is("path")) | | |
| 368 | .path | | |
| 369 | else if (inf.is("linkpath")) | | |
| 370 | .linkpath | | |
| 371 | else if (inf.is("size")) | | |
| 372 | .size | | |
| 373 | else { | | |
| 374 | try self.advance(inf.value_off + inf.value_len); | | |
| 375 | try self.checkAttributeEnding(); | | |
| 376 | continue; | | |
| 377 | }; | | |
| 378 | try self.advance(inf.value_off); // position reader at the start of the value | | |
| 379 | return PaxAttribute{ .key = key, .value_len = inf.value_len, .parent = self }; | | |
| 380 | } | | |
| 381 | } | | |
| 382 | | | |
| 383 | fn checkAttributeEnding(self: *PaxFileReader) !void { | | |
| 384 | if (self.reader.readByte() != '\n') return error.InvalidPaxAttribute; | | |
| 385 | try self.advance(1); | | |
| 386 | } | | |
| 387 | | | |
| 388 | fn advance(self: *PaxFileReader, len: usize) !void { | | |
| 389 | self.offset += len; | | |
| 390 | try self.reader.skip(len); | | |
| 391 | } | | |
| 392 | }; | | |
| 393 | }; | | |
| 394 | } | | |
| 395 | | | |
| 396 | fn Iterator(comptime BufferedReaderType: type) type { | | |
| 397 | return struct { | | |
| 398 | // scratch buffer for file attributes | | |
| 399 | scratch: struct { | | |
| 400 | // size: two paths (name and link_name) and files size bytes (24 in pax attribute) | | |
| 401 | buffer: [std.fs.MAX_PATH_BYTES * 2 + 24]u8 = undefined, | | |
| 402 | tail: usize = 0, | | |
| 403 | | | |
| 404 | name: []const u8 = undefined, | | |
| 405 | link_name: []const u8 = undefined, | | |
| 406 | size: usize = 0, | | |
| 407 | | | |
| 408 | // Allocate size of the buffer for some attribute. | | |
| 409 | fn alloc(self: *@This(), size: usize) ![]u8 { | | |
| 410 | const free_size = self.buffer.len - self.tail; | | |
| 411 | if (size > free_size) return error.TarScratchBufferOverflow; | | |
| 412 | const head = self.tail; | | |
| 413 | self.tail += size; | | |
| 414 | assert(self.tail <= self.buffer.len); | | |
| 415 | return self.buffer[head..self.tail]; | | |
| 416 | } | | |
| 417 | | | |
| 418 | // Reset buffer and all fields. | | |
| 419 | fn reset(self: *@This()) void { | | |
| 420 | self.tail = 0; | | |
| 421 | self.name = self.buffer[0..0]; | | |
| 422 | self.link_name = self.buffer[0..0]; | | |
| 423 | self.size = 0; | | |
| 424 | } | | |
| 425 | | | |
| 426 | fn append(self: *@This(), header: Header) !void { | | |
| 427 | if (self.size == 0) self.size = try header.fileSize(); | | |
| 428 | if (self.link_name.len == 0) { | | |
| 429 | const link_name = header.linkName(); | | |
| 430 | if (link_name.len > 0) { | | |
| 431 | const buf = try self.alloc(link_name.len); | | |
| 432 | @memcpy(buf, link_name); | | |
| 433 | self.link_name = buf; | | |
| 434 | } | | |
| 435 | } | | |
| 436 | if (self.name.len == 0) { | | |
| 437 | self.name = try header.fullName((try self.alloc(MAX_HEADER_NAME_SIZE))[0..MAX_HEADER_NAME_SIZE]); | | |
| 438 | } | | |
| 439 | } | | |
| 440 | } = .{}, | | |
| 441 | | | |
| 442 | reader: BufferedReaderType, | | |
| 443 | diagnostics: ?*Options.Diagnostics, | | |
| 444 | padding: usize = 0, // bytes of padding to the end of the block | | |
| 445 | | | |
| 446 | const Self = @This(); | | |
| 447 | | | |
| 448 | pub const File = struct { | | |
| 449 | name: []const u8, // name of file, symlink or directory | | |
| 450 | link_name: []const u8, // target name of symlink | | |
| 451 | size: usize, // size of the file in bytes | | |
| 452 | mode: u32, | | |
| 453 | file_type: Header.FileType, | | |
| 454 | | | |
| 455 | reader: *BufferedReaderType, | | |
| 456 | | | |
| 457 | // Writes file content to writer. | | |
| 458 | pub fn write(self: File, writer: anytype) !void { | | |
| 459 | try self.reader.write(writer, self.size); | | |
| 460 | } | | |
| 461 | | | |
| 462 | // Skips file content. Advances reader. | | |
| 463 | pub fn skip(self: File) !void { | | |
| 464 | try self.reader.skip(self.size); | | |
| 465 | } | | |
| 466 | }; | | |
| 467 | | | |
| 468 | // Externally, `next` iterates through the tar archive as if it is a | | |
| 469 | // series of files. Internally, the tar format often uses fake "files" | | |
| 470 | // to add meta data that describes the next file. These meta data | | |
| 471 | // "files" should not normally be visible to the outside. As such, this | | |
| 472 | // loop iterates through one or more "header files" until it finds a | | |
| 473 | // "normal file". | | |
| 474 | pub fn next(self: *Self) !?File { | | |
| 475 | self.scratch.reset(); | | |
| 476 | | | |
| 477 | while (try self.reader.readHeader(self.padding)) |block_bytes| { | | |
| 478 | const header = Header{ .bytes = block_bytes[0..BLOCK_SIZE] }; | | |
| 479 | if (try header.checkChksum() == 0) return null; // zero block found | | |
| 480 | | | |
| 481 | const file_type = header.fileType(); | | |
| 482 | const size: usize = @intCast(try header.fileSize()); | | |
| 483 | self.padding = blockPadding(size); | | |
| 484 | | | |
| 485 | switch (file_type) { | | |
| 486 | // File types to retrun upstream | | |
| 487 | .directory, .normal, .symbolic_link => { | | |
| 488 | try self.scratch.append(header); | | |
| 489 | const file = File{ | | |
| 490 | .file_type = file_type, | | |
| 491 | .name = self.scratch.name, | | |
| 492 | .link_name = self.scratch.link_name, | | |
| 493 | .size = self.scratch.size, | | |
| 494 | .reader = &self.reader, | | |
| 495 | .mode = try header.mode(), | | |
| 496 | }; | | |
| 497 | self.padding = blockPadding(file.size); | | |
| 498 | return file; | | |
| 499 | }, | | |
| 500 | // Prefix header types | | |
| 501 | .gnu_long_name => { | | |
| 502 | self.scratch.name = nullStr(try self.reader.copy(try self.scratch.alloc(size))); | | |
| 503 | }, | | |
| 504 | .gnu_long_link => { | | |
| 505 | self.scratch.link_name = nullStr(try self.reader.copy(try self.scratch.alloc(size))); | | |
| 506 | }, | | |
| 507 | .extended_header => { | | |
| 508 | if (size == 0) continue; | | |
| 509 | // Use just attributes from last extended header. | | |
| 510 | self.scratch.reset(); | | |
| 511 | | | |
| 512 | var rdr = self.reader.paxFileReader(size); | | |
| 513 | while (try rdr.next()) |attr| { | | |
| 514 | switch (attr.key) { | | |
| 515 | .path => { | | |
| 516 | self.scratch.name = try noNull(try attr.value(try self.scratch.alloc(attr.value_len))); | | |
| 517 | }, | | |
| 518 | .linkpath => { | | |
| 519 | self.scratch.link_name = try noNull(try attr.value(try self.scratch.alloc(attr.value_len))); | | |
| 520 | }, | | |
| 521 | .size => { | | |
| 522 | self.scratch.size = try std.fmt.parseInt(usize, try attr.value(try self.scratch.alloc(attr.value_len)), 10); | | |
| 523 | }, | | |
| 524 | } | | |
| 525 | } | | |
| 526 | }, | | |
| 527 | // Ignored header type | | |
| 528 | .global_extended_header => { | | |
| 529 | self.reader.skip(size) catch return error.TarHeadersTooBig; | | |
| 530 | }, | | |
| 531 | // All other are unsupported header types | | |
| 532 | else => { | | |
| 533 | const d = self.diagnostics orelse return error.TarUnsupportedFileType; | | |
| 534 | try d.errors.append(d.allocator, .{ .unsupported_file_type = .{ | | |
| 535 | .file_name = try d.allocator.dupe(u8, header.name()), | | |
| 536 | .file_type = file_type, | | |
| 537 | } }); | | |
| 538 | }, | | |
| 539 | } | | |
| 540 | } | | |
| 541 | return null; | | |
| 542 | } | | |
| 543 | }; | | |
| 544 | } | | |
| 545 | | | |
| 546 | pub fn iterator(underlying_reader: anytype, diagnostics: ?*Options.Diagnostics) Iterator(BufferedReader(@TypeOf(underlying_reader))) { | | |
| 547 | return .{ | | |
| 548 | .reader = bufferedReader(underlying_reader), | | |
| 549 | .diagnostics = diagnostics, | | |
| 550 | }; | | |
| 551 | } | | |
| 552 | | | |
| 553 | fn bufferedReader(underlying_reader: anytype) BufferedReader(@TypeOf(underlying_reader)) { | | |
| 554 | return BufferedReader(@TypeOf(underlying_reader)){ | | |
| 555 | .underlying_reader = underlying_reader, | | |
| 556 | }; | | |
| 557 | } | | |
| 558 | | | |
| 559 | pub fn pipeToFileSystem(dir: std.fs.Dir, reader: anytype, options: Options) !void { | 228 | pub fn pipeToFileSystem(dir: std.fs.Dir, reader: anytype, options: Options) !void { |
| 560 | switch (options.mode_mode) { | 229 | switch (options.mode_mode) { |
| 561 | .ignore => {}, | 230 | .ignore => {}, |
| ... | @@ -569,7 +238,7 @@ pub fn pipeToFileSystem(dir: std.fs.Dir, reader: anytype, options: Options) !voi | ... | @@ -569,7 +238,7 @@ pub fn pipeToFileSystem(dir: std.fs.Dir, reader: anytype, options: Options) !voi |
| 569 | }, | 238 | }, |
| 570 | } | 239 | } |
| 571 | | 240 | |
| 572 | var iter = iterator(reader, options.diagnostics); | 241 | var iter = tarReader(reader, options.diagnostics); |
| 573 | | 242 | |
| 574 | while (try iter.next()) |file| { | 243 | while (try iter.next()) |file| { |
| 575 | switch (file.file_type) { | 244 | switch (file.file_type) { |
| ... | @@ -662,82 +331,37 @@ test "tar stripComponents" { | ... | @@ -662,82 +331,37 @@ test "tar stripComponents" { |
| 662 | try expectEqualStrings("c", try stripComponents("a/b/c", 2)); | 331 | try expectEqualStrings("c", try stripComponents("a/b/c", 2)); |
| 663 | } | 332 | } |
| 664 | | 333 | |
| 665 | const PaxAttributeInfo = struct { | | |
| 666 | size: usize, | | |
| 667 | key: []const u8, | | |
| 668 | value_off: usize, | | |
| 669 | value_len: usize, | | |
| 670 | | | |
| 671 | inline fn is(self: @This(), key: []const u8) bool { | | |
| 672 | return (std.mem.eql(u8, self.key, key)); | | |
| 673 | } | | |
| 674 | }; | | |
| 675 | | | |
| 676 | fn parsePaxAttribute(data: []const u8, max_size: usize) !PaxAttributeInfo { | | |
| 677 | const pos_space = std.mem.indexOfScalar(u8, data, ' ') orelse return error.InvalidPaxAttribute; | | |
| 678 | const pos_equals = std.mem.indexOfScalarPos(u8, data, pos_space, '=') orelse return error.InvalidPaxAttribute; | | |
| 679 | const kv_size = try std.fmt.parseInt(usize, data[0..pos_space], 10); | | |
| 680 | if (kv_size > max_size or kv_size < pos_equals + 2) { | | |
| 681 | return error.InvalidPaxAttribute; | | |
| 682 | } | | |
| 683 | const key = data[pos_space + 1 .. pos_equals]; | | |
| 684 | return .{ | | |
| 685 | .size = kv_size, | | |
| 686 | .key = try noNull(key), | | |
| 687 | .value_off = pos_equals + 1, | | |
| 688 | .value_len = kv_size - pos_equals - 2, | | |
| 689 | }; | | |
| 690 | } | | |
| 691 | | | |
| 692 | fn noNull(str: []const u8) ![]const u8 { | 334 | fn noNull(str: []const u8) ![]const u8 { |
| 693 | if (std.mem.indexOfScalar(u8, str, 0)) |_| return error.InvalidPaxAttribute; | 335 | if (std.mem.indexOfScalar(u8, str, 0)) |_| return error.InvalidPaxAttribute; |
| 694 | return str; | 336 | return str; |
| 695 | } | 337 | } |
| 696 | | 338 | |
| 697 | test "tar parsePaxAttribute" { | 339 | test "tar run Go test cases" { |
| 698 | const expectEqual = std.testing.expectEqual; | 340 | const Case = struct { |
| 699 | const expectEqualStrings = std.testing.expectEqualStrings; | 341 | const File = struct { |
| 700 | const expectError = std.testing.expectError; | 342 | name: []const u8, |
| 701 | const prefix = "1011 path="; | 343 | size: usize = 0, |
| 702 | const file_name = "0123456789" ** 100; | 344 | mode: u32 = 0, |
| 703 | const header = prefix ++ file_name ++ "\n"; | 345 | link_name: []const u8 = &[0]u8{}, |
| 704 | const attr_info = try parsePaxAttribute(header, 1011); | 346 | file_type: Header.FileType = .normal, |
| 705 | try expectEqual(@as(usize, 1011), attr_info.size); | 347 | truncated: bool = false, // when there is no file body, just header, usefull for huge files |
| 706 | try expectEqualStrings("path", attr_info.key); | 348 | }; |
| 707 | try expectEqual(prefix.len, attr_info.value_off); | | |
| 708 | try expectEqual(file_name.len, attr_info.value_len); | | |
| 709 | try expectEqual(attr_info, try parsePaxAttribute(header, 1012)); | | |
| 710 | try expectError(error.InvalidPaxAttribute, parsePaxAttribute(header, 1010)); | | |
| 711 | try expectError(error.InvalidPaxAttribute, parsePaxAttribute("", 0)); | | |
| 712 | try expectError(error.InvalidPaxAttribute, parsePaxAttribute("13 pa\x00th=abc\n", 1024)); // null in key | | |
| 713 | } | | |
| 714 | | 349 | |
| 715 | const TestCase = struct { | 350 | path: []const u8, // path to the tar archive file on dis |
| 716 | const File = struct { | 351 | files: []const File = &[_]@This().File{}, // expected files to found in archive |
| 717 | name: []const u8, | 352 | chksums: []const []const u8 = &[_][]const u8{}, // chksums of files content |
| 718 | size: usize = 0, | 353 | err: ?anyerror = null, // parsing should fail with this error |
| 719 | mode: u32 = 0, | | |
| 720 | link_name: []const u8 = &[0]u8{}, | | |
| 721 | file_type: Header.FileType = .normal, | | |
| 722 | truncated: bool = false, // when there is no file body, just header, usefull for huge files | | |
| 723 | }; | 354 | }; |
| 724 | | 355 | |
| 725 | path: []const u8, // path to the tar archive file on dis | | |
| 726 | files: []const File = &[_]TestCase.File{}, // expected files to found in archive | | |
| 727 | chksums: []const []const u8 = &[_][]const u8{}, // chksums of files content | | |
| 728 | err: ?anyerror = null, // parsing should fail with this error | | |
| 729 | }; | | |
| 730 | | | |
| 731 | test "tar run Go test cases" { | | |
| 732 | const test_dir = if (std.os.getenv("GO_TAR_TESTDATA_PATH")) |path| | 356 | const test_dir = if (std.os.getenv("GO_TAR_TESTDATA_PATH")) |path| |
| 733 | try std.fs.openDirAbsolute(path, .{}) | 357 | try std.fs.openDirAbsolute(path, .{}) |
| 734 | else | 358 | else |
| 735 | return error.SkipZigTest; | 359 | return error.SkipZigTest; |
| 736 | | 360 | |
| 737 | const cases = [_]TestCase{ | 361 | const cases = [_]Case{ |
| 738 | .{ | 362 | .{ |
| 739 | .path = "gnu.tar", | 363 | .path = "gnu.tar", |
| 740 | .files = &[_]TestCase.File{ | 364 | .files = &[_]Case.File{ |
| 741 | .{ | 365 | .{ |
| 742 | .name = "small.txt", | 366 | .name = "small.txt", |
| 743 | .size = 5, | 367 | .size = 5, |
| ... | @@ -760,7 +384,7 @@ test "tar run Go test cases" { | ... | @@ -760,7 +384,7 @@ test "tar run Go test cases" { |
| 760 | }, | 384 | }, |
| 761 | .{ | 385 | .{ |
| 762 | .path = "star.tar", | 386 | .path = "star.tar", |
| 763 | .files = &[_]TestCase.File{ | 387 | .files = &[_]Case.File{ |
| 764 | .{ | 388 | .{ |
| 765 | .name = "small.txt", | 389 | .name = "small.txt", |
| 766 | .size = 5, | 390 | .size = 5, |
| ... | @@ -779,7 +403,7 @@ test "tar run Go test cases" { | ... | @@ -779,7 +403,7 @@ test "tar run Go test cases" { |
| 779 | }, | 403 | }, |
| 780 | .{ | 404 | .{ |
| 781 | .path = "v7.tar", | 405 | .path = "v7.tar", |
| 782 | .files = &[_]TestCase.File{ | 406 | .files = &[_]Case.File{ |
| 783 | .{ | 407 | .{ |
| 784 | .name = "small.txt", | 408 | .name = "small.txt", |
| 785 | .size = 5, | 409 | .size = 5, |
| ... | @@ -798,7 +422,7 @@ test "tar run Go test cases" { | ... | @@ -798,7 +422,7 @@ test "tar run Go test cases" { |
| 798 | }, | 422 | }, |
| 799 | .{ | 423 | .{ |
| 800 | .path = "pax.tar", | 424 | .path = "pax.tar", |
| 801 | .files = &[_]TestCase.File{ | 425 | .files = &[_]Case.File{ |
| 802 | .{ | 426 | .{ |
| 803 | .name = "a/123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100", | 427 | .name = "a/123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100", |
| 804 | .size = 7, | 428 | .size = 7, |
| ... | @@ -824,7 +448,7 @@ test "tar run Go test cases" { | ... | @@ -824,7 +448,7 @@ test "tar run Go test cases" { |
| 824 | .{ | 448 | .{ |
| 825 | // size is in pax attribute | 449 | // size is in pax attribute |
| 826 | .path = "pax-pos-size-file.tar", | 450 | .path = "pax-pos-size-file.tar", |
| 827 | .files = &[_]TestCase.File{ | 451 | .files = &[_]Case.File{ |
| 828 | .{ | 452 | .{ |
| 829 | .name = "foo", | 453 | .name = "foo", |
| 830 | .size = 999, | 454 | .size = 999, |
| ... | @@ -839,7 +463,7 @@ test "tar run Go test cases" { | ... | @@ -839,7 +463,7 @@ test "tar run Go test cases" { |
| 839 | .{ | 463 | .{ |
| 840 | // has pax records which we are not interested in | 464 | // has pax records which we are not interested in |
| 841 | .path = "pax-records.tar", | 465 | .path = "pax-records.tar", |
| 842 | .files = &[_]TestCase.File{ | 466 | .files = &[_]Case.File{ |
| 843 | .{ | 467 | .{ |
| 844 | .name = "file", | 468 | .name = "file", |
| 845 | }, | 469 | }, |
| ... | @@ -848,7 +472,7 @@ test "tar run Go test cases" { | ... | @@ -848,7 +472,7 @@ test "tar run Go test cases" { |
| 848 | .{ | 472 | .{ |
| 849 | // has global records which we are ignoring | 473 | // has global records which we are ignoring |
| 850 | .path = "pax-global-records.tar", | 474 | .path = "pax-global-records.tar", |
| 851 | .files = &[_]TestCase.File{ | 475 | .files = &[_]Case.File{ |
| 852 | .{ | 476 | .{ |
| 853 | .name = "file1", | 477 | .name = "file1", |
| 854 | }, | 478 | }, |
| ... | @@ -865,7 +489,7 @@ test "tar run Go test cases" { | ... | @@ -865,7 +489,7 @@ test "tar run Go test cases" { |
| 865 | }, | 489 | }, |
| 866 | .{ | 490 | .{ |
| 867 | .path = "nil-uid.tar", | 491 | .path = "nil-uid.tar", |
| 868 | .files = &[_]TestCase.File{ | 492 | .files = &[_]Case.File{ |
| 869 | .{ | 493 | .{ |
| 870 | .name = "P1050238.JPG.log", | 494 | .name = "P1050238.JPG.log", |
| 871 | .size = 14, | 495 | .size = 14, |
| ... | @@ -880,7 +504,7 @@ test "tar run Go test cases" { | ... | @@ -880,7 +504,7 @@ test "tar run Go test cases" { |
| 880 | .{ | 504 | .{ |
| 881 | // has xattrs and pax records which we are ignoring | 505 | // has xattrs and pax records which we are ignoring |
| 882 | .path = "xattrs.tar", | 506 | .path = "xattrs.tar", |
| 883 | .files = &[_]TestCase.File{ | 507 | .files = &[_]Case.File{ |
| 884 | .{ | 508 | .{ |
| 885 | .name = "small.txt", | 509 | .name = "small.txt", |
| 886 | .size = 5, | 510 | .size = 5, |
| ... | @@ -901,7 +525,7 @@ test "tar run Go test cases" { | ... | @@ -901,7 +525,7 @@ test "tar run Go test cases" { |
| 901 | }, | 525 | }, |
| 902 | .{ | 526 | .{ |
| 903 | .path = "gnu-multi-hdrs.tar", | 527 | .path = "gnu-multi-hdrs.tar", |
| 904 | .files = &[_]TestCase.File{ | 528 | .files = &[_]Case.File{ |
| 905 | .{ | 529 | .{ |
| 906 | .name = "GNU2/GNU2/long-path-name", | 530 | .name = "GNU2/GNU2/long-path-name", |
| 907 | .link_name = "GNU4/GNU4/long-linkpath-name", | 531 | .link_name = "GNU4/GNU4/long-linkpath-name", |
| ... | @@ -917,7 +541,7 @@ test "tar run Go test cases" { | ... | @@ -917,7 +541,7 @@ test "tar run Go test cases" { |
| 917 | .{ | 541 | .{ |
| 918 | // should use values only from last pax header | 542 | // should use values only from last pax header |
| 919 | .path = "pax-multi-hdrs.tar", | 543 | .path = "pax-multi-hdrs.tar", |
| 920 | .files = &[_]TestCase.File{ | 544 | .files = &[_]Case.File{ |
| 921 | .{ | 545 | .{ |
| 922 | .name = "bar", | 546 | .name = "bar", |
| 923 | .link_name = "PAX4/PAX4/long-linkpath-name", | 547 | .link_name = "PAX4/PAX4/long-linkpath-name", |
| ... | @@ -927,7 +551,7 @@ test "tar run Go test cases" { | ... | @@ -927,7 +551,7 @@ test "tar run Go test cases" { |
| 927 | }, | 551 | }, |
| 928 | .{ | 552 | .{ |
| 929 | .path = "gnu-long-nul.tar", | 553 | .path = "gnu-long-nul.tar", |
| 930 | .files = &[_]TestCase.File{ | 554 | .files = &[_]Case.File{ |
| 931 | .{ | 555 | .{ |
| 932 | .name = "0123456789", | 556 | .name = "0123456789", |
| 933 | .mode = 0o644, | 557 | .mode = 0o644, |
| ... | @@ -936,7 +560,7 @@ test "tar run Go test cases" { | ... | @@ -936,7 +560,7 @@ test "tar run Go test cases" { |
| 936 | }, | 560 | }, |
| 937 | .{ | 561 | .{ |
| 938 | .path = "gnu-utf8.tar", | 562 | .path = "gnu-utf8.tar", |
| 939 | .files = &[_]TestCase.File{ | 563 | .files = &[_]Case.File{ |
| 940 | .{ | 564 | .{ |
| 941 | .name = "☺☻☹☺☻☹☺☻☹☺☻☹☺☻☹☺☻☹☺☻☹☺☻☹☺☻☹☺☻☹☺☻☹☺☻☹☺☻☹☺☻☹☺☻☹☺☻☹☺☻☹☺☻☹", | 565 | .name = "☺☻☹☺☻☹☺☻☹☺☻☹☺☻☹☺☻☹☺☻☹☺☻☹☺☻☹☺☻☹☺☻☹☺☻☹☺☻☹☺☻☹☺☻☹☺☻☹☺☻☹☺☻☹", |
| 942 | .mode = 0o644, | 566 | .mode = 0o644, |
| ... | @@ -945,7 +569,7 @@ test "tar run Go test cases" { | ... | @@ -945,7 +569,7 @@ test "tar run Go test cases" { |
| 945 | }, | 569 | }, |
| 946 | .{ | 570 | .{ |
| 947 | .path = "gnu-not-utf8.tar", | 571 | .path = "gnu-not-utf8.tar", |
| 948 | .files = &[_]TestCase.File{ | 572 | .files = &[_]Case.File{ |
| 949 | .{ | 573 | .{ |
| 950 | .name = "hi\x80\x81\x82\x83bye", | 574 | .name = "hi\x80\x81\x82\x83bye", |
| 951 | .mode = 0o644, | 575 | .mode = 0o644, |
| ... | @@ -980,7 +604,7 @@ test "tar run Go test cases" { | ... | @@ -980,7 +604,7 @@ test "tar run Go test cases" { |
| 980 | .{ | 604 | .{ |
| 981 | // has magic with space at end instead of null | 605 | // has magic with space at end instead of null |
| 982 | .path = "invalid-go17.tar", | 606 | .path = "invalid-go17.tar", |
| 983 | .files = &[_]TestCase.File{ | 607 | .files = &[_]Case.File{ |
| 984 | .{ | 608 | .{ |
| 985 | .name = "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa/foo", | 609 | .name = "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa/foo", |
| 986 | }, | 610 | }, |
| ... | @@ -988,7 +612,7 @@ test "tar run Go test cases" { | ... | @@ -988,7 +612,7 @@ test "tar run Go test cases" { |
| 988 | }, | 612 | }, |
| 989 | .{ | 613 | .{ |
| 990 | .path = "ustar-file-devs.tar", | 614 | .path = "ustar-file-devs.tar", |
| 991 | .files = &[_]TestCase.File{ | 615 | .files = &[_]Case.File{ |
| 992 | .{ | 616 | .{ |
| 993 | .name = "file", | 617 | .name = "file", |
| 994 | .mode = 0o644, | 618 | .mode = 0o644, |
| ... | @@ -997,7 +621,7 @@ test "tar run Go test cases" { | ... | @@ -997,7 +621,7 @@ test "tar run Go test cases" { |
| 997 | }, | 621 | }, |
| 998 | .{ | 622 | .{ |
| 999 | .path = "trailing-slash.tar", | 623 | .path = "trailing-slash.tar", |
| 1000 | .files = &[_]TestCase.File{ | 624 | .files = &[_]Case.File{ |
| 1001 | .{ | 625 | .{ |
| 1002 | .name = "123456789/" ** 30, | 626 | .name = "123456789/" ** 30, |
| 1003 | .file_type = .directory, | 627 | .file_type = .directory, |
| ... | @@ -1007,7 +631,7 @@ test "tar run Go test cases" { | ... | @@ -1007,7 +631,7 @@ test "tar run Go test cases" { |
| 1007 | .{ | 631 | .{ |
| 1008 | // Has size in gnu extended format. To represent size bigger than 8 GB. | 632 | // Has size in gnu extended format. To represent size bigger than 8 GB. |
| 1009 | .path = "writer-big.tar", | 633 | .path = "writer-big.tar", |
| 1010 | .files = &[_]TestCase.File{ | 634 | .files = &[_]Case.File{ |
| 1011 | .{ | 635 | .{ |
| 1012 | .name = "tmp/16gig.txt", | 636 | .name = "tmp/16gig.txt", |
| 1013 | .size = 16 * 1024 * 1024 * 1024, | 637 | .size = 16 * 1024 * 1024 * 1024, |
| ... | @@ -1019,7 +643,7 @@ test "tar run Go test cases" { | ... | @@ -1019,7 +643,7 @@ test "tar run Go test cases" { |
| 1019 | .{ | 643 | .{ |
| 1020 | // Size in gnu extended format, and name in pax attribute. | 644 | // Size in gnu extended format, and name in pax attribute. |
| 1021 | .path = "writer-big-long.tar", | 645 | .path = "writer-big-long.tar", |
| 1022 | .files = &[_]TestCase.File{ | 646 | .files = &[_]Case.File{ |
| 1023 | .{ | 647 | .{ |
| 1024 | .name = "longname/" ** 15 ++ "16gig.txt", | 648 | .name = "longname/" ** 15 ++ "16gig.txt", |
| 1025 | .size = 16 * 1024 * 1024 * 1024, | 649 | .size = 16 * 1024 * 1024 * 1024, |
| ... | @@ -1034,7 +658,8 @@ test "tar run Go test cases" { | ... | @@ -1034,7 +658,8 @@ test "tar run Go test cases" { |
| 1034 | var fs_file = try test_dir.openFile(case.path, .{}); | 658 | var fs_file = try test_dir.openFile(case.path, .{}); |
| 1035 | defer fs_file.close(); | 659 | defer fs_file.close(); |
| 1036 | | 660 | |
| 1037 | var iter = iterator(fs_file.reader(), null); | 661 | //var iter = iterator(fs_file.reader(), null); |
| | 662 | var iter = tarReader(fs_file.reader(), null); |
| 1038 | var i: usize = 0; | 663 | var i: usize = 0; |
| 1039 | while (iter.next() catch |err| { | 664 | while (iter.next() catch |err| { |
| 1040 | if (case.err) |e| { | 665 | if (case.err) |e| { |
| ... | @@ -1072,6 +697,10 @@ const Md5Writer = struct { | ... | @@ -1072,6 +697,10 @@ const Md5Writer = struct { |
| 1072 | self.h.update(buf); | 697 | self.h.update(buf); |
| 1073 | } | 698 | } |
| 1074 | | 699 | |
| | 700 | pub fn writeByte(self: *Md5Writer, byte: u8) !void { |
| | 701 | self.h.update(&[_]u8{byte}); |
| | 702 | } |
| | 703 | |
| 1075 | pub fn chksum(self: *Md5Writer) [32]u8 { | 704 | pub fn chksum(self: *Md5Writer) [32]u8 { |
| 1076 | var s = [_]u8{0} ** 16; | 705 | var s = [_]u8{0} ** 16; |
| 1077 | self.h.final(&s); | 706 | self.h.final(&s); |
| ... | @@ -1079,19 +708,113 @@ const Md5Writer = struct { | ... | @@ -1079,19 +708,113 @@ const Md5Writer = struct { |
| 1079 | } | 708 | } |
| 1080 | }; | 709 | }; |
| 1081 | | 710 | |
| 1082 | test "tar PaxFileReader" { | 711 | fn paxReader(reader: anytype, size: usize) PaxReader(@TypeOf(reader)) { |
| 1083 | const Attribute = struct { | 712 | return PaxReader(@TypeOf(reader)){ |
| 1084 | const PaxKeyKind = enum { | 713 | .reader = reader, |
| 1085 | path, | 714 | .size = size, |
| 1086 | linkpath, | 715 | }; |
| 1087 | size, | 716 | } |
| | 717 | |
| | 718 | const PaxAttrKind = enum { |
| | 719 | path, |
| | 720 | linkpath, |
| | 721 | size, |
| | 722 | }; |
| | 723 | |
| | 724 | fn PaxReader(comptime ReaderType: type) type { |
| | 725 | return struct { |
| | 726 | size: usize, |
| | 727 | reader: ReaderType, |
| | 728 | |
| | 729 | const Self = @This(); |
| | 730 | |
| | 731 | const Attr = struct { |
| | 732 | kind: PaxAttrKind, |
| | 733 | len: usize, |
| | 734 | reader: ReaderType, |
| | 735 | |
| | 736 | // Copies pax attribute value into destination buffer. |
| | 737 | // Must be called with destination buffer of size at least value_len. |
| | 738 | pub fn value(self: Attr, dst: []u8) ![]const u8 { |
| | 739 | assert(self.len <= dst.len); |
| | 740 | const buf = dst[0..self.len]; |
| | 741 | const n = try self.reader.readAll(buf); |
| | 742 | if (n < self.len) return error.UnexpectedEndOfStream; |
| | 743 | try checkRecordEnd(self.reader); |
| | 744 | return noNull(buf); |
| | 745 | } |
| 1088 | }; | 746 | }; |
| 1089 | key: PaxKeyKind, | 747 | |
| 1090 | value: []const u8, | 748 | // Iterates over pax records. Returns known records. Caller has to call |
| | 749 | // value in Record, to advance reader across value. |
| | 750 | pub fn next(self: *Self) !?Attr { |
| | 751 | var buf: [128]u8 = undefined; |
| | 752 | var fbs = std.io.fixedBufferStream(&buf); |
| | 753 | |
| | 754 | // An extended header consists of one or more records, each constructed as follows: |
| | 755 | // "%d %s=%s\n", <length>, <keyword>, <value> |
| | 756 | while (self.size > 0) { |
| | 757 | fbs.reset(); |
| | 758 | // read length |
| | 759 | try self.reader.streamUntilDelimiter(fbs.writer(), ' ', null); |
| | 760 | const rec_len = try std.fmt.parseInt(usize, fbs.getWritten(), 10); // record len in bytes |
| | 761 | var pos = try fbs.getPos() + 1; // bytes used for record len + separator |
| | 762 | fbs.reset(); |
| | 763 | // read keyword |
| | 764 | try self.reader.streamUntilDelimiter(fbs.writer(), '=', null); |
| | 765 | const keyword = fbs.getWritten(); |
| | 766 | pos += try fbs.getPos() + 1; // keyword bytes + separator |
| | 767 | try checkKeyword(keyword); |
| | 768 | // get value_len |
| | 769 | if (rec_len < pos + 1) return error.InvalidPaxAttribute; |
| | 770 | const value_len = rec_len - pos - 1; // pos = start of value, -1 => without \n record terminator |
| | 771 | |
| | 772 | self.size -= rec_len; |
| | 773 | const kind: PaxAttrKind = if (eql(keyword, "path")) |
| | 774 | .path |
| | 775 | else if (eql(keyword, "linkpath")) |
| | 776 | .linkpath |
| | 777 | else if (eql(keyword, "size")) |
| | 778 | .size |
| | 779 | else { |
| | 780 | try self.reader.skipBytes(value_len, .{}); |
| | 781 | try checkRecordEnd(self.reader); |
| | 782 | continue; |
| | 783 | }; |
| | 784 | return Attr{ |
| | 785 | .kind = kind, |
| | 786 | .len = value_len, |
| | 787 | .reader = self.reader, |
| | 788 | }; |
| | 789 | } |
| | 790 | |
| | 791 | return null; |
| | 792 | } |
| | 793 | |
| | 794 | inline fn eql(a: []const u8, b: []const u8) bool { |
| | 795 | return std.mem.eql(u8, a, b); |
| | 796 | } |
| | 797 | |
| | 798 | fn checkKeyword(keyword: []const u8) !void { |
| | 799 | if (std.mem.indexOfScalar(u8, keyword, 0)) |_| return error.InvalidPaxAttribute; |
| | 800 | } |
| | 801 | |
| | 802 | // Checks that each record ends with new line. |
| | 803 | fn checkRecordEnd(reader: ReaderType) !void { |
| | 804 | if (try reader.readByte() != '\n') return error.InvalidPaxAttribute; |
| | 805 | } |
| | 806 | }; |
| | 807 | } |
| | 808 | |
| | 809 | test "tar PaxReader" { |
| | 810 | const Attr = struct { |
| | 811 | kind: PaxAttrKind, |
| | 812 | value: []const u8 = undefined, |
| | 813 | err: ?anyerror = null, |
| 1091 | }; | 814 | }; |
| 1092 | const cases = [_]struct { | 815 | const cases = [_]struct { |
| 1093 | data: []const u8, | 816 | data: []const u8, |
| 1094 | attrs: []const Attribute, | 817 | attrs: []const Attr, |
| 1095 | err: ?anyerror = null, | 818 | err: ?anyerror = null, |
| 1096 | }{ | 819 | }{ |
| 1097 | .{ // valid but unknown keys | 820 | .{ // valid but unknown keys |
| ... | @@ -1103,7 +826,7 @@ test "tar PaxFileReader" { | ... | @@ -1103,7 +826,7 @@ test "tar PaxFileReader" { |
| 1103 | \\9 a=name | 826 | \\9 a=name |
| 1104 | \\ | 827 | \\ |
| 1105 | , | 828 | , |
| 1106 | .attrs = &[_]Attribute{}, | 829 | .attrs = &[_]Attr{}, |
| 1107 | }, | 830 | }, |
| 1108 | .{ // mix of known and unknown keys | 831 | .{ // mix of known and unknown keys |
| 1109 | .data = | 832 | .data = |
| ... | @@ -1115,10 +838,10 @@ test "tar PaxFileReader" { | ... | @@ -1115,10 +838,10 @@ test "tar PaxFileReader" { |
| 1115 | \\13 key2=val2 | 838 | \\13 key2=val2 |
| 1116 | \\ | 839 | \\ |
| 1117 | , | 840 | , |
| 1118 | .attrs = &[_]Attribute{ | 841 | .attrs = &[_]Attr{ |
| 1119 | .{ .key = .path, .value = "name" }, | 842 | .{ .kind = .path, .value = "name" }, |
| 1120 | .{ .key = .linkpath, .value = "link" }, | 843 | .{ .kind = .linkpath, .value = "link" }, |
| 1121 | .{ .key = .size, .value = "123" }, | 844 | .{ .kind = .size, .value = "123" }, |
| 1122 | }, | 845 | }, |
| 1123 | }, | 846 | }, |
| 1124 | .{ // too short size of the second key-value pair | 847 | .{ // too short size of the second key-value pair |
| ... | @@ -1127,8 +850,8 @@ test "tar PaxFileReader" { | ... | @@ -1127,8 +850,8 @@ test "tar PaxFileReader" { |
| 1127 | \\10 linkpath=value | 850 | \\10 linkpath=value |
| 1128 | \\ | 851 | \\ |
| 1129 | , | 852 | , |
| 1130 | .attrs = &[_]Attribute{ | 853 | .attrs = &[_]Attr{ |
| 1131 | .{ .key = .path, .value = "name" }, | 854 | .{ .kind = .path, .value = "name" }, |
| 1132 | }, | 855 | }, |
| 1133 | .err = error.InvalidPaxAttribute, | 856 | .err = error.InvalidPaxAttribute, |
| 1134 | }, | 857 | }, |
| ... | @@ -1136,36 +859,237 @@ test "tar PaxFileReader" { | ... | @@ -1136,36 +859,237 @@ test "tar PaxFileReader" { |
| 1136 | .data = | 859 | .data = |
| 1137 | \\13 path=name | 860 | \\13 path=name |
| 1138 | \\19 linkpath=value | 861 | \\19 linkpath=value |
| | 862 | \\6 k=1 |
| 1139 | \\ | 863 | \\ |
| 1140 | , | 864 | , |
| 1141 | .attrs = &[_]Attribute{ | 865 | .attrs = &[_]Attr{ |
| 1142 | .{ .key = .path, .value = "name" }, | 866 | .{ .kind = .path, .value = "name" }, |
| | 867 | .{ .kind = .linkpath, .err = error.InvalidPaxAttribute }, |
| | 868 | }, |
| | 869 | }, |
| | 870 | .{ // null in keyword is not valid |
| | 871 | .data = "13 path=name\n" ++ "7 k\x00b=1\n", |
| | 872 | .attrs = &[_]Attr{ |
| | 873 | .{ .kind = .path, .value = "name" }, |
| 1143 | }, | 874 | }, |
| 1144 | .err = error.InvalidPaxAttribute, | 875 | .err = error.InvalidPaxAttribute, |
| 1145 | }, | 876 | }, |
| | 877 | .{ // null in value is not valid |
| | 878 | .data = "23 path=name\x00with null\n", |
| | 879 | .attrs = &[_]Attr{ |
| | 880 | .{ .kind = .path, .err = error.InvalidPaxAttribute }, |
| | 881 | }, |
| | 882 | }, |
| | 883 | .{ // 1000 characters path |
| | 884 | .data = "1011 path=" ++ "0123456789" ** 100 ++ "\n", |
| | 885 | .attrs = &[_]Attr{ |
| | 886 | .{ .kind = .path, .value = "0123456789" ** 100 }, |
| | 887 | }, |
| | 888 | }, |
| 1146 | }; | 889 | }; |
| 1147 | var buffer: [1024]u8 = undefined; | 890 | var buffer: [1024]u8 = undefined; |
| 1148 | | 891 | |
| 1149 | for (cases) |case| { | 892 | outer: for (cases) |case| { |
| 1150 | var stream = std.io.fixedBufferStream(case.data); | 893 | var stream = std.io.fixedBufferStream(case.data); |
| 1151 | var brdr = bufferedReader(stream.reader()); | 894 | var rdr = paxReader(stream.reader(), case.data.len); |
| 1152 | | 895 | |
| 1153 | var rdr = brdr.paxFileReader(case.data.len); | | |
| 1154 | var i: usize = 0; | 896 | var i: usize = 0; |
| 1155 | while (rdr.next() catch |err| { | 897 | while (rdr.next() catch |err| { |
| 1156 | if (case.err) |e| { | 898 | if (case.err) |e| { |
| 1157 | try std.testing.expectEqual(e, err); | 899 | try std.testing.expectEqual(e, err); |
| 1158 | continue; | 900 | continue; |
| 1159 | } else { | | |
| 1160 | return err; | | |
| 1161 | } | 901 | } |
| | 902 | return err; |
| 1162 | }) |attr| : (i += 1) { | 903 | }) |attr| : (i += 1) { |
| 1163 | try std.testing.expectEqualStrings( | 904 | const exp = case.attrs[i]; |
| 1164 | case.attrs[i].value, | 905 | try std.testing.expectEqual(exp.kind, attr.kind); |
| 1165 | try attr.value(&buffer), | 906 | const value = attr.value(&buffer) catch |err| { |
| 1166 | ); | 907 | if (exp.err) |e| { |
| | 908 | try std.testing.expectEqual(e, err); |
| | 909 | break :outer; |
| | 910 | } |
| | 911 | return err; |
| | 912 | }; |
| | 913 | try std.testing.expectEqualStrings(exp.value, value); |
| 1167 | } | 914 | } |
| 1168 | try std.testing.expectEqual(case.attrs.len, i); | 915 | try std.testing.expectEqual(case.attrs.len, i); |
| 1169 | try std.testing.expect(case.err == null); | 916 | try std.testing.expect(case.err == null); |
| 1170 | } | 917 | } |
| 1171 | } | 918 | } |
| | 919 | |
| | 920 | pub fn tarReader(reader: anytype, diagnostics: ?*Options.Diagnostics) TarReader(@TypeOf(reader)) { |
| | 921 | return .{ |
| | 922 | .reader = reader, |
| | 923 | .diagnostics = diagnostics, |
| | 924 | }; |
| | 925 | } |
| | 926 | |
| | 927 | fn TarReader(comptime ReaderType: type) type { |
| | 928 | return struct { |
| | 929 | // scratch buffer for file attributes |
| | 930 | scratch: struct { |
| | 931 | // size: two paths (name and link_name) and files size bytes (24 in pax attribute) |
| | 932 | buffer: [std.fs.MAX_PATH_BYTES * 2 + 24]u8 = undefined, |
| | 933 | tail: usize = 0, |
| | 934 | |
| | 935 | name: []const u8 = undefined, |
| | 936 | link_name: []const u8 = undefined, |
| | 937 | size: usize = 0, |
| | 938 | |
| | 939 | // Allocate size of the buffer for some attribute. |
| | 940 | fn alloc(self: *@This(), size: usize) ![]u8 { |
| | 941 | const free_size = self.buffer.len - self.tail; |
| | 942 | if (size > free_size) return error.TarScratchBufferOverflow; |
| | 943 | const head = self.tail; |
| | 944 | self.tail += size; |
| | 945 | assert(self.tail <= self.buffer.len); |
| | 946 | return self.buffer[head..self.tail]; |
| | 947 | } |
| | 948 | |
| | 949 | // Reset buffer and all fields. |
| | 950 | fn reset(self: *@This()) void { |
| | 951 | self.tail = 0; |
| | 952 | self.name = self.buffer[0..0]; |
| | 953 | self.link_name = self.buffer[0..0]; |
| | 954 | self.size = 0; |
| | 955 | } |
| | 956 | |
| | 957 | fn append(self: *@This(), header: Header) !void { |
| | 958 | if (self.size == 0) self.size = try header.fileSize(); |
| | 959 | if (self.link_name.len == 0) { |
| | 960 | const link_name = header.linkName(); |
| | 961 | if (link_name.len > 0) { |
| | 962 | const buf = try self.alloc(link_name.len); |
| | 963 | @memcpy(buf, link_name); |
| | 964 | self.link_name = buf; |
| | 965 | } |
| | 966 | } |
| | 967 | if (self.name.len == 0) { |
| | 968 | self.name = try header.fullName((try self.alloc(MAX_HEADER_NAME_SIZE))[0..MAX_HEADER_NAME_SIZE]); |
| | 969 | } |
| | 970 | } |
| | 971 | } = .{}, |
| | 972 | |
| | 973 | reader: ReaderType, |
| | 974 | diagnostics: ?*Options.Diagnostics, |
| | 975 | padding: usize = 0, // bytes of padding to the end of the block |
| | 976 | header_buffer: [BLOCK_SIZE]u8 = undefined, |
| | 977 | |
| | 978 | const Self = @This(); |
| | 979 | |
| | 980 | pub const File = struct { |
| | 981 | name: []const u8, // name of file, symlink or directory |
| | 982 | link_name: []const u8, // target name of symlink |
| | 983 | size: usize, // size of the file in bytes |
| | 984 | mode: u32, |
| | 985 | file_type: Header.FileType, |
| | 986 | |
| | 987 | reader: *ReaderType, |
| | 988 | |
| | 989 | // Writes file content to writer. |
| | 990 | pub fn write(self: File, writer: anytype) !void { |
| | 991 | var n = self.size; |
| | 992 | while (n > 0) : (n -= 1) { |
| | 993 | const byte: u8 = try self.reader.readByte(); |
| | 994 | try writer.writeByte(byte); |
| | 995 | } |
| | 996 | } |
| | 997 | |
| | 998 | // Skips file content. Advances reader. |
| | 999 | pub fn skip(self: File) !void { |
| | 1000 | try self.reader.skipBytes(self.size, .{}); |
| | 1001 | } |
| | 1002 | }; |
| | 1003 | |
| | 1004 | fn readHeader(self: *Self) !?Header { |
| | 1005 | if (self.padding > 0) { |
| | 1006 | try self.reader.skipBytes(self.padding, .{}); |
| | 1007 | } |
| | 1008 | const n = try self.reader.readAll(&self.header_buffer); |
| | 1009 | if (n == 0) return null; |
| | 1010 | if (n < BLOCK_SIZE) return error.UnexpectedEndOfStream; |
| | 1011 | const header = Header{ .bytes = self.header_buffer[0..BLOCK_SIZE] }; |
| | 1012 | if (try header.checkChksum() == 0) return null; |
| | 1013 | return header; |
| | 1014 | } |
| | 1015 | |
| | 1016 | fn readString(self: *Self, size: usize) ![]const u8 { |
| | 1017 | const buf = try self.scratch.alloc(size); |
| | 1018 | try self.reader.readNoEof(buf); |
| | 1019 | return nullStr(buf); |
| | 1020 | } |
| | 1021 | |
| | 1022 | // Externally, `next` iterates through the tar archive as if it is a |
| | 1023 | // series of files. Internally, the tar format often uses fake "files" |
| | 1024 | // to add meta data that describes the next file. These meta data |
| | 1025 | // "files" should not normally be visible to the outside. As such, this |
| | 1026 | // loop iterates through one or more "header files" until it finds a |
| | 1027 | // "normal file". |
| | 1028 | pub fn next(self: *Self) !?File { |
| | 1029 | self.scratch.reset(); |
| | 1030 | |
| | 1031 | while (try self.readHeader()) |header| { |
| | 1032 | const file_type = header.fileType(); |
| | 1033 | const size: usize = @intCast(try header.fileSize()); |
| | 1034 | self.padding = blockPadding(size); |
| | 1035 | |
| | 1036 | switch (file_type) { |
| | 1037 | // File types to retrun upstream |
| | 1038 | .directory, .normal, .symbolic_link => { |
| | 1039 | try self.scratch.append(header); |
| | 1040 | const file = File{ |
| | 1041 | .file_type = file_type, |
| | 1042 | .name = self.scratch.name, |
| | 1043 | .link_name = self.scratch.link_name, |
| | 1044 | .size = self.scratch.size, |
| | 1045 | .reader = &self.reader, |
| | 1046 | .mode = try header.mode(), |
| | 1047 | }; |
| | 1048 | self.padding = blockPadding(file.size); |
| | 1049 | return file; |
| | 1050 | }, |
| | 1051 | // Prefix header types |
| | 1052 | .gnu_long_name => { |
| | 1053 | self.scratch.name = try self.readString(size); |
| | 1054 | }, |
| | 1055 | .gnu_long_link => { |
| | 1056 | self.scratch.link_name = try self.readString(size); |
| | 1057 | }, |
| | 1058 | .extended_header => { |
| | 1059 | if (size == 0) continue; |
| | 1060 | // Use just attributes from last extended header. |
| | 1061 | self.scratch.reset(); |
| | 1062 | |
| | 1063 | var rdr = paxReader(self.reader, size); |
| | 1064 | while (try rdr.next()) |attr| { |
| | 1065 | switch (attr.kind) { |
| | 1066 | .path => { |
| | 1067 | self.scratch.name = try attr.value(try self.scratch.alloc(attr.len)); |
| | 1068 | }, |
| | 1069 | .linkpath => { |
| | 1070 | self.scratch.link_name = try attr.value(try self.scratch.alloc(attr.len)); |
| | 1071 | }, |
| | 1072 | .size => { |
| | 1073 | self.scratch.size = try std.fmt.parseInt(usize, try attr.value(try self.scratch.alloc(attr.len)), 10); |
| | 1074 | }, |
| | 1075 | } |
| | 1076 | } |
| | 1077 | }, |
| | 1078 | // Ignored header type |
| | 1079 | .global_extended_header => { |
| | 1080 | self.reader.skipBytes(size, .{}) catch return error.TarHeadersTooBig; |
| | 1081 | }, |
| | 1082 | // All other are unsupported header types |
| | 1083 | else => { |
| | 1084 | const d = self.diagnostics orelse return error.TarUnsupportedFileType; |
| | 1085 | try d.errors.append(d.allocator, .{ .unsupported_file_type = .{ |
| | 1086 | .file_name = try d.allocator.dupe(u8, header.name()), |
| | 1087 | .file_type = file_type, |
| | 1088 | } }); |
| | 1089 | }, |
| | 1090 | } |
| | 1091 | } |
| | 1092 | return null; |
| | 1093 | } |
| | 1094 | }; |
| | 1095 | } |