lib/suitar: allow implicitly-sized entries
diff --git a/doc/spec/suitar-spec.md b/doc/spec/suitar-spec.md index 1af3bae..e714e69 100644 --- a/doc/spec/suitar-spec.md +++ b/doc/spec/suitar-spec.md
@@ -14,8 +14,9 @@ or [NIE](./nie-spec.md) is to image files: an uncompressed, "designed for Unix pipes" format that is trivial to read or write in a few hundred lines of code, ideally in a memory-safe programming language. It's a format for what the -Chromium web browser's "Rule of 2" security advice calls -[https://chromium.googlesource.com/chromium/src/+/master/docs/security/rule-of-2.md#normalization](Normalization). +Chromium web browser's +[https://chromium.googlesource.com/chromium/src/+/master/docs/security/rule-of-2.md](Rule of 2) +security advice calls Normalization. ## Subset of TAR @@ -37,7 +38,7 @@ For example, answering "are these two file names duplicates (and does the destination FS / OS accept or reject duplicates)?" may depend on the -_destination_ FS / OS's case-sensitivity and Unicode Normalization +_destination_ FS / OS's case-sensitivity and Unicode normalization configuration (and whether case folding and normalization elides DICPs, Unicode Default-Ignorable Code Points), not the _source_ SUITAR archive per se. @@ -51,9 +52,9 @@ manual](https://ftp.gnu.org/old-gnu/Manuals/tar-1.12/html_node/tar_123.html)), SUITAR has further restrictions: -- Entries are either regular files (`REGTYPE`), sparse files (`GNUTYPE_SPARSE`) - or directories (`DIRTYPE`). There is no support for hard links, symlinks, - device files or other non-standard files. +- Entries are either regular files (`REGTYPE` or `AREGTYPE`), sparse files + (`GNUTYPE_SPARSE`) or directories (`DIRTYPE`). There is no support for hard + links, symlinks, device files or other non-standard files. - Sparse files must be completely sparse. Their content must be one contiguous span of NUL (zero) bytes that covers the entire file. - File and directory names must obey the "File Name Validity" rules, below. @@ -73,16 +74,6 @@ TAR" or "the USTAR extensions to TAR", other than what's implied by the subset of the GNU extensions that SUITAR explicitly uses. -Encoders have no meaningful choices, bar one exception. There is only one valid -SUITAR encoding (unlike full TAR's backwards-compatible choice between base-8 -or base-256 encoding of various sufficiently small numbers) for any given file -or directory entry (its combination of type, name, size, mode, modTime and -contents). - -The one exception is that, if a file's contents are all NUL bytes (including -zero-sized files), an encoder can choose between a `REGTYPE` regular file (with -explicit NULs) or a `GNUTYPE_SPARSE` sparse file (with implicit NULs). - ### File Name Validity @@ -118,12 +109,70 @@ - 1 `GNUTYPE_LONGNAME` header block. - 1 or more payload blocks containing the file or directory name. -- 1 `REGTYPE`, `GNUTYPE_SPARSE` or `DIRTYPE` header block. +- 1 `REGTYPE`, `AREGTYPE`, `GNUTYPE_SPARSE` or `DIRTYPE` header block. - If `REGTYPE`, 0 or more payload blocks containing the file contents. -- If not `REGTYPE`, no further blocks. +- If `AREGTYPE`, see "Implicitly-Sized Files" below. +- If neither `REGTYPE` or `AREGTYPE`, no further blocks. -### Header Blocks +## Implicitly-Sized Files + +For TAR itself, each entry's contents is preceded by the entry's header that +states the entire contents' size in bytes. SUITAR keeps that formal structure +but also uses additional convention to represent entries whose size is only +known at the end, not the start, of the entry. + +For example, when decompressing `foobar.dat.gz` from a GZIP stream to SUITAR +(an archive with one entry: `foobar.dat`), the uncompressed `foobar.dat` size +is not known until the end of the GZIP-formatted input stream is reached. + +For historical reasons, TAR itself has two typeflag codes for regular files +(`REGTYPE` and `AREGTYPE`) and both codes are largely equivalent. SUITAR, by +convention, treats them differently: `REGTYPE` is for the common case, where +the contents' size is known up-front, and `AREGTYPE` means the contents that +follow are partial and to be continued. + +An implicitly-sized file is partitioned into 2 or more chunks. The final chunk +is `REGTYPE` and all other chunks are `AREGTYPE`. Each chunk's size is explicit +(and zero is a valid size) but the number of chunks isn't known until the final +`REGTYPE` chunk is delivered. The entry's structure is: + +- 1 `GNUTYPE_LONGNAME` header block. +- 1 or more payload blocks containing the file or directory name. +- 1 or more non-final chunks, each being: + - 1 `AREGTYPE` header block, stating the chunk contents' size. + - 0 or more payload blocks containing the chunk contents. +- 1 final chunk, being: + - 1 `REGTYPE` header block, stating the chunk contents' size. + - 0 or more payload blocks containing the chunk contents. + +"0 non-final chunks" is actually valid in some sense, equivalent to an +explicitly-sized `REGTYPE` entry (of exactly one chunk). + +The file name still comes from the `GNUTYPE_LONGNAME` payload. There is only +one `GNUTYPE_LONGNAME` header block, not one per chunk. + +Other metadata (mode and modTime, but not file size) comes from the initial +chunk's header block. For non-initial chunks, the mode must be `"644"` and the +modTime must be zero. + + +### Implicitly-Sized Fallback Behavior + +Other TAR-reading tools and libraries, which do not understand SUITAR's +"implicitly-sized files" convention, will fall back to treating all non-initial +chunks as separate regular files. These will all have the same file name +(`"\x13sUItAR"`, due to the SUITAR magic signature), which does not satisfy the +File Name Validity rules, but this fallback name will not be presented by +decoders that understand the convention. + +SUITAR encoders are discouraged from using this implicitly-sized files +convention unless the surrounding context ensures that the decoders also speak +SUITAR, not just TAR per se. This can be more likely if writing SUITAR over a +Unix pipe (with a known program on the other end), compared to writing to disk. + + +## Header Blocks Like all blocks, each header block is 512 bytes long. Each header block also starts with a 12-byte magic signature (that is not valid UTF-8), identifying @@ -165,6 +214,7 @@ modTime as an 8-byte big-endian `uint64`, 6-byte checksum (see below) or 1-byte type, which must be one of: +- `'\x00'` for `AREGTYPE`. - `'0'` for `REGTYPE`. - `'5'` for `DIRTYPE`, in which case mode must be `"755"` and physical size must be all zeroes. @@ -190,17 +240,20 @@ ### Header Checksum -A 512-byte header block's checksum value is simply the sum of each byte (after -converting from `uint8` to `uint32`, to avoid overflow) in the block, at -offsets in the two half-open ranges `0 .. 148` and `156 .. 512`, which excludes -the 8 bytes for the 6-byte checksum itself plus another two hard-coded bytes -`"\x00\x20"`. +A 512-byte header block's checksum value is simply 256 plus the sum of each +byte (after converting from `uint8` to `uint32`, to avoid overflow) in the +block, at offsets in the two half-open ranges `0 .. 148` and `156 .. 512`, +which excludes the 8 bytes for the 6-byte checksum itself plus another two +hard-coded bytes `"\x00\x20"`. + +That "256 plus" is equivalent to summing over the entire `0 .. 512` range if +valuing the 8 checksum bytes in the range `148 .. 156` as being `'\x20'`. That checksum value is written as a 6-byte ASCII octal number in the header. For example, `4853` (decimal) would be encoded as `"011365"` (octal). -### Payload Blocks +## Payload Blocks Each entry has one or more payload blocks, between its two header blocks, containing the file or directory name. The name length (including a trailing @@ -210,10 +263,13 @@ to a multiple of 512 gives the number of 512-byte payload blocks that contain the name. All padding bytes in the name's final payload block must be NUL. -For `REGTYPE` entries, the second header block's physical size value gives the -reconstructed file's size and rounding that up to a multiple of 512 gives the -number of 512-byte payload blocks that contain the file contents. Again, all -padding bytes in the contents' final payload block must be NUL. +For `REGTYPE` or `AREGTYPE` entries, the name payload is followed by one or +more chunks. Concatenating the chunks' contents reconstructs the file. Only the +final chunk is `REGTYPE` and all others are `AREGTYPE`. Each chunk has one +header block and zero or more payload blocks. The header block's physical size +value gives the chunk's size and rounding that up to a multiple of 512 gives +the number of 512-byte payload blocks that contain the chunk contents. Again, +all padding bytes in the contents' final payload block must be NUL. For other entries (`DIRTYPE` or `GNUTYPE_SPARSE`), there are no further payload blocks after the second header block. @@ -222,7 +278,7 @@ gives the reconstructed file's size and its contents are all NUL bytes. -# Reference Implementation +## Reference Implementation The [google/wuffs](https://github.com/google/wuffs) repository, which holds this specification document, also holds a
diff --git a/lib/suitar/suitar.go b/lib/suitar/suitar.go index 9ad3dda..7582639 100644 --- a/lib/suitar/suitar.go +++ b/lib/suitar/suitar.go
@@ -38,6 +38,7 @@ errClosed = errors.New("suitar: closed") errHeaderSize = errors.New("suitar: inconsistent Header.Size and Write length") errHeaderTypeflag = errors.New("suitar: inconsistent Header.Typeflag for Write") + errImplicitSize = errors.New("suitar: implicit size is too large") errWriteANonNul = errors.New("suitar: Write a non-NUL byte to a sparse entry") ) @@ -96,6 +97,7 @@ TypeGNUSparse = 'S' // Sparse file (its contents are all NUL bytes). typeGNULongName = 'L' + typeRegA = '\x00' ) // Valid values for Header.Mode. The zero value is invalid. @@ -104,6 +106,15 @@ Mode755 = int64(0o755) // "rwxr-xr-x" mode bits, also known as permission bits. ) +// SizeIsImplicit can be passed in the Writer.WriteHeader's argument's +// Header.Size field to mean that the contents' size-in-bytes is not yet known, +// but is implied by the total length of the []byte passed to Writer.Write, up +// until the next call to Writer.WriteHeader or Writer.Close. +// +// When reading, the actual size is returned by Reader.NumBytesRead. Call this +// after Reader.Read has returned io.EOF and before the next Reader.Next call. +const SizeIsImplicit = int64(-1) + // IsValidHeaderName returns whether name is a valid Header.Name field value. func IsValidHeaderName(name string) bool { return (len(name) < 4096) && @@ -248,7 +259,8 @@ // - Typeflag must be one of three values (TypeReg, TypeDir, TypeGNUSparse). // In particular, it cannot be zero. // - Name must satisfy IsValidHeaderName. -// - Size must be non-negative. It must be 0 if Typeflag is TypeDir. +// - Size must be non-negative or, if Typeflag is TypeReg, it can also be -1, +// the value of SizeIsImplicit. It must be 0 if Typeflag is TypeDir. // - Mode must be one of two values (Mode644, Mode755). It must be Mode755 if // Typeflag is TypeDir. // - ModTime.Unix() must be a non-negative int64. In particular, ModTime must @@ -275,9 +287,14 @@ } } + size := h.Size + if (size == -1) && (h.Typeflag == TypeReg) { + size = 0 + } + const maxExcl = 1 << 53 m := h.ModTime.Unix() - return (0 <= h.Size) && (h.Size < maxExcl) && + return (0 <= size) && (size < maxExcl) && (0 <= m) && (m < maxExcl) && IsValidHeaderName(h.Name) } @@ -292,7 +309,7 @@ err error w io.Writer header Header - remaining int64 + remaining int64 // If negative, ^remaining is the accumulated implicit size. bIndex int32 block block nameBuf [4096]byte @@ -301,9 +318,29 @@ func (w *Writer) flush() error { if w.err != nil { return w.err + + } else if w.remaining < 0 { + initBlock(&w.block, TypeReg, int64(w.bIndex), Mode644, 0) + if _, err := w.w.Write(w.block[:]); err != nil { + w.err = err + return w.err + } + + if w.bIndex != 0 { + end := (w.bIndex + 511) &^ 511 + clear(w.nameBuf[w.bIndex:end]) + if _, err := w.w.Write(w.nameBuf[:end]); err != nil { + w.err = err + return w.err + } + w.bIndex = 0 + } + return nil + } else if (w.remaining != 0) && (w.header.Typeflag != TypeGNUSparse) { w.err = errHeaderSize return w.err + } else if w.bIndex != 0 { clear(w.block[w.bIndex:]) if _, err := w.w.Write(w.block[:]); err != nil { @@ -378,6 +415,9 @@ physicalSize := size if typeflag == TypeGNUSparse { physicalSize = 0 + } else if size < 0 { + physicalSize = 0 + typeflag = typeRegA } setI64(b, 0x07C, physicalSize) @@ -411,13 +451,15 @@ return 0, w.err } - tooMuch := int64(len(b)) > w.remaining + tooMuch := (int64(len(b)) > w.remaining) && (w.remaining >= 0) if tooMuch { b = b[:w.remaining] } ret := 0 - if w.header.Typeflag == TypeReg { + if w.remaining < 0 { + ret, w.err = w.writeImplicitlySized(b) + } else if w.header.Typeflag == TypeReg { ret, w.err = w.writeReg(b) } else if w.header.Typeflag == TypeGNUSparse { ret, w.err = w.writeSparse(b) @@ -434,6 +476,62 @@ return ret, w.err } +func (w *Writer) writeImplicitlySized(b []byte) (int, error) { + if len(b) == 0 { + return 0, nil + } + + const maxExcl = 1 << 53 + if n := int64(len(b)); (n >= maxExcl) || ((n + ^w.remaining) >= maxExcl) { + return 0, errImplicitSize + } + + if (w.bIndex > 0) || (len(b) < (len(w.nameBuf) - int(w.bIndex))) { + n := copy(w.nameBuf[w.bIndex:], b) + w.bIndex += int32(n) + b = b[n:] + if len(b) == 0 { + return n, nil + } + } + + ret := 0 + + split := len(b) &^ 511 + prefix, suffix := b[:split], b[split:] + + if size := int64(w.bIndex) + int64(len(prefix)); size > 0 { + initBlock(&w.block, typeRegA, size, Mode644, 0) + if _, err := w.w.Write(w.block[:]); err != nil { + return ret, err + } + } + + if w.bIndex > 0 { + n, err := w.w.Write(w.nameBuf[:w.bIndex]) + ret += n + if err != nil { + return ret, err + } + w.bIndex = 0 + } + + if len(prefix) > 0 { + n, err := w.w.Write(prefix) + ret += n + if err != nil { + return ret, err + } + } + + if len(suffix) > 0 { + w.bIndex = int32(copy(w.nameBuf[:], suffix)) + ret += len(suffix) + } + + return ret, nil +} + func (w *Writer) writeReg(b []byte) (int, error) { if len(b) == 0 { return 0, nil @@ -508,18 +606,21 @@ // Reader provides sequential reading of a SUITAR archive. type Reader struct { - err error - r io.Reader - remaining int64 - numPadding int32 - sparse bool - block block - nameBuf [4096]byte + err error + r io.Reader + remaining int64 + numBytesRead int64 + numPadding int32 + sparse bool + sizeIsImplicit bool + block block + nameBuf [4096]byte } // Next advances to the next entry in the SUITAR archive, preparing to read the // file's contents (as r is also an io.Reader). func (r *Reader) Next() (Header, error) { + r.numBytesRead = 0 if r.err != nil { return Header{}, r.err } else if r.remaining > 0 { @@ -532,7 +633,7 @@ if _, err := readFullNoEOF(r.r, r.block[:]); err != nil { r.err = err return Header{}, r.err - } else if r.block[0x09C] == 0 { + } else if r.block[0] == 0 { // SUITAR ends with 2 blocks (1024 bytes) of zeroes. if !isAllZeroes(r.block[:]) { r.err = errBadHeader @@ -579,6 +680,11 @@ r.remaining = size r.numPadding = int32(roundUp512(uint64(size)) - uint64(size)) r.sparse = typeflag == TypeGNUSparse + r.sizeIsImplicit = typeflag == typeRegA + if r.sizeIsImplicit { + typeflag = TypeReg + size = SizeIsImplicit + } return Header{ Typeflag: typeflag, @@ -611,7 +717,7 @@ } typeflag := b[0x09C] - if (typeflag != TypeReg) && (typeflag != TypeDir) && (typeflag != TypeGNUSparse) { + if (typeflag != TypeReg) && (typeflag != TypeDir) && (typeflag != TypeGNUSparse) && (typeflag != typeRegA) { return 0, 0, 0, 0, errBadHeader } @@ -649,15 +755,45 @@ // Read satisfies io.Reader. func (r *Reader) Read(b []byte) (int, error) { + n, err := r.read(b) + r.numBytesRead += int64(n) + return n, err +} + +func (r *Reader) read(b []byte) (int, error) { if r.err != nil { return 0, r.err - } else if r.remaining == 0 { - return 0, io.EOF - } else if len(b) == 0 { - return 0, nil + } + + for r.remaining == 0 { + if !r.sizeIsImplicit { + return 0, io.EOF + } + if _, err := readFullNoEOF(r.r, r.block[:0x200]); err != nil { + r.err = err + return 0, err + } + typeflag, size, mode, modTime, err := parseBlock1(&r.block) + if err != nil { + r.err = err + return 0, err + } else if ((typeflag != TypeReg) && (typeflag != typeRegA)) || (mode != Mode644) || (modTime != 0) { + r.err = errBadHeader + return 0, err + } else if (r.numBytesRead + size) >= (1 << 53) { + r.err = errImplicitSize + return 0, err + } + r.remaining = size + r.numPadding = int32(roundUp512(uint64(size)) - uint64(size)) + r.sizeIsImplicit = typeflag == typeRegA } b = b[:int(min(r.remaining, int64(len(b))))] + if len(b) == 0 { + return 0, nil + } + if r.sparse { n := len(b) clear(b) @@ -688,7 +824,11 @@ r.numPadding = 0 } if (err == nil) || (err == io.EOF) { - return n, io.EOF + if r.sizeIsImplicit { + return n, nil + } else { + return n, io.EOF + } } } else if err == io.EOF { @@ -698,3 +838,9 @@ r.err = err return n, r.err } + +// NumBytesRead returns the number of bytes read by all previous Reader.Read +// calls. The counter resets to zero on each Reader.Next call. +func (r *Reader) NumBytesRead() int64 { + return r.numBytesRead +}
diff --git a/lib/suitar/suitar_test.go b/lib/suitar/suitar_test.go index d3d4513..c6eade7 100644 --- a/lib/suitar/suitar_test.go +++ b/lib/suitar/suitar_test.go
@@ -17,6 +17,7 @@ "fmt" "hash/crc32" "io" + "math/rand" "os" "reflect" "testing" @@ -32,8 +33,12 @@ return len(b), nil } -func testWriter(tt *testing.T, sparse bool) { - f, err := os.Open("../../test/data/archive.tar") +func testWriter(tt *testing.T, sparse bool, implicit bool) { + srcFilename := "../../test/data/archive.tar" + if implicit { + srcFilename = "../../test/data/various-peacocks.dense.suitar" + } + f, err := os.Open(srcFilename) if err != nil { tt.Fatalf("os.Open: %v", err) } @@ -51,10 +56,15 @@ tt.Fatalf("Next: %v", err) } + size := tHeader.Size + if implicit { + size = SizeIsImplicit + } + sHeader := &Header{ Typeflag: tHeader.Typeflag, Name: tHeader.Name, - Size: tHeader.Size, + Size: size, Mode: tHeader.Mode, ModTime: tHeader.ModTime, } @@ -79,11 +89,11 @@ } got := buf.Bytes() - wantFilename := "../../test/data/archive" + wantFilename := "../../test/data/archive.dense.suitar" if sparse { - wantFilename += ".sparse.suitar" - } else { - wantFilename += ".dense.suitar" + wantFilename = "../../test/data/archive.sparse.suitar" + } else if implicit { + wantFilename = "../../test/data/various-peacocks.implicit.suitar" } want, err := os.ReadFile(wantFilename) if err != nil { @@ -181,8 +191,9 @@ } } -func TestWriterDense(tt *testing.T) { testWriter(tt, false) } -func TestWriterSparse(tt *testing.T) { testWriter(tt, true) } +func TestWriterDense(tt *testing.T) { testWriter(tt, false, false) } +func TestWriterSparse(tt *testing.T) { testWriter(tt, true, false) } +func TestWriterImplicit(tt *testing.T) { testWriter(tt, false, true) } func TestReaderDenseCheck(tt *testing.T) { testReader(tt, false, false) } func TestReaderDenseIgnore(tt *testing.T) { testReader(tt, false, true) } func TestReaderSparseCheck(tt *testing.T) { testReader(tt, true, false) } @@ -424,3 +435,82 @@ func TestWriteTooMuchRegular(tt *testing.T) { testWriteTooMuch(tt, TypeReg) } func TestWriteTooMuchSparse(tt *testing.T) { testWriteTooMuch(tt, TypeGNUSparse) } + +func TestSizeIsImplicit(tt *testing.T) { + pi, err := os.ReadFile("../../test/data/pi.txt") + if err != nil { + tt.Fatalf("ReadFile: %v", err) + } + + writeSizes := []int{ + 1, 2, 3, 4, + 10, 20, 50, 100, + 250, 251, 252, 253, + 254, 255, 256, 257, + 401, 402, 403, 404, + 510, 511, 512, 513, + 1024, 1234, 4094, 4095, + 4096, 4097, 4098, -1, + } + + totalSizes := []int(nil) + + buf := bytes.Buffer{} + sWriter := NewWriter(&buf) + + rng := rand.New(rand.NewSource(0)) + for i := range 100 { + if err := sWriter.WriteHeader(&Header{ + Typeflag: TypeReg, + Name: fmt.Sprintf("%03d.dat", i), + Size: SizeIsImplicit, + Mode: Mode644, + ModTime: time.Unix(0, 0), + }); err != nil { + tt.Fatalf("WriteHeader: %v", err) + } + totalSize := 0 + + for range 20 { + writeSize := min(writeSizes[rng.Intn(len(writeSizes))], len(pi)-totalSize) + if writeSize <= 0 { + break + } + if _, err := sWriter.Write(pi[totalSize : totalSize+writeSize]); err != nil { + tt.Fatalf("Write: %v", err) + } + totalSize += writeSize + } + + totalSizes = append(totalSizes, totalSize) + } + + if err := sWriter.Close(); err != nil { + tt.Fatalf("Close: %v", err) + } + + sReader := NewReader(&buf) + for i, totalSize := range totalSizes { + sHeader, err := sReader.Next() + if err == io.EOF { + break + } else if err != nil { + tt.Fatalf("Next: %v", err) + } else if sHeader.Size != SizeIsImplicit { + tt.Fatalf("sHeader.Size: got %d, want %d", sHeader.Size, SizeIsImplicit) + } else if want := fmt.Sprintf("%03d.dat", i); sHeader.Name != want { + tt.Fatalf("sHeader.Name: got %q, want %q", sHeader.Name, want) + } + + got, err := io.ReadAll(sReader) + if err != nil { + tt.Fatalf("ReadAll: %v", err) + } else if len(got) != totalSize { + tt.Fatalf("bytes read: got %d, want %d", len(got), totalSize) + } else if !bytes.Equal(got, pi[:totalSize]) { + tt.Fatalf("bytes read: contents differ") + } else if n := sReader.NumBytesRead(); n != int64(totalSize) { + tt.Fatalf("NumBytesRead: got %d, want %d", n, totalSize) + } + } +}
diff --git a/test/data/various-peacocks.dense.suitar b/test/data/various-peacocks.dense.suitar new file mode 100644 index 0000000..7f3af1c --- /dev/null +++ b/test/data/various-peacocks.dense.suitar Binary files differ
diff --git a/test/data/various-peacocks.implicit.suitar b/test/data/various-peacocks.implicit.suitar new file mode 100644 index 0000000..a9bcd87 --- /dev/null +++ b/test/data/various-peacocks.implicit.suitar Binary files differ