perf(pages): skip per-file hashes during package inspect

Inspect deployment archives via file handles and declared sizes instead of loading the whole package and hashing every member, while keeping whole-package checksums for Agent integrity checks.
This commit is contained in:
ryan
2026-07-17 18:37:34 +08:00
parent 285f127d48
commit 3b9f4daa4e
7 changed files with 236 additions and 56 deletions
+6 -14
View File
@@ -4,8 +4,6 @@
package pagesarchive
import (
"crypto/sha256"
"encoding/hex"
"fmt"
"io"
"math"
@@ -25,8 +23,10 @@ type Limits struct {
// FileEntry is a regular file discovered inside a deployment package.
type FileEntry struct {
Path string
Size int64
Path string
Size int64
// Checksum is retained for API/schema compatibility and is left empty.
// Integrity is enforced via the whole-package SHA-256 on the deployment record.
Checksum string
}
@@ -45,9 +45,10 @@ type Entry struct {
IsDir bool
// IsSymlink marks symbolic links (unsupported for Pages).
IsSymlink bool
// Size is the declared uncompressed size when known; 0 means unknown.
// Size is the declared uncompressed size when known; 0 means empty or unknown.
Size uint64
// Open returns a reader for the entry body. Caller must Close it.
// May be unavailable for inspect-only tar listings (body not materialized).
Open func() (io.ReadCloser, error)
}
@@ -79,15 +80,6 @@ func copyLimited(dst io.Writer, src io.Reader, declaredSize uint64, maxBytes int
return written, err
}
func checksumReader(src io.Reader, declaredSize uint64, maxBytes int64) (string, int64, error) {
hash := sha256.New()
written, err := copyLimited(hash, src, declaredSize, maxBytes)
if err != nil {
return "", written, err
}
return hex.EncodeToString(hash.Sum(nil)), written, nil
}
func writeEntryFile(targetPath string, src io.Reader, declaredSize uint64, maxBytes int64, perm os.FileMode) (int64, error) {
if err := os.MkdirAll(filepath.Dir(targetPath), dirPerm); err != nil {
return 0, err
+29 -4
View File
@@ -4,7 +4,9 @@
package pagesarchive
import (
"bytes"
"fmt"
"io"
"os"
"path/filepath"
)
@@ -31,20 +33,43 @@ func ExtractBytes(data []byte, format Format, destDir string, opts ExtractOption
return err
}
}
entries, err := listEntries(data, format)
entries, err := listEntriesAt(bytes.NewReader(data), int64(len(data)), format, true)
if err != nil {
return err
}
return extractEntries(entries, destDir, opts)
}
// ExtractFile reads path and extracts it into destDir.
// ExtractFile opens path and extracts it into destDir without buffering the
// whole archive as an intermediate []byte for zip/7z (ReaderAt). Tar-family
// formats still materialize member bodies so random Open works for extract.
func ExtractFile(filePath string, format Format, destDir string, opts ExtractOptions) error {
data, err := os.ReadFile(filePath) //nolint:gosec // controlled path
file, err := os.Open(filePath) //nolint:gosec // controlled path
if err != nil {
return err
}
return ExtractBytes(data, format, destDir, opts)
defer func() { _ = file.Close() }()
info, err := file.Stat()
if err != nil {
return err
}
if format == "" {
head := make([]byte, 512)
n, readErr := file.ReadAt(head, 0)
if readErr != nil && readErr != io.EOF {
return readErr
}
format, err = DetectFormat(filePath, head[:n])
if err != nil {
return err
}
}
entries, err := listEntriesAt(file, info.Size(), format, true)
if err != nil {
return err
}
return extractEntries(entries, destDir, opts)
}
func extractEntries(entries []Entry, destDir string, opts ExtractOptions) error {
+68 -11
View File
@@ -4,7 +4,10 @@
package pagesarchive
import (
"bytes"
"fmt"
"io"
"math"
"os"
"path"
"strings"
@@ -18,15 +21,38 @@ type InspectOptions struct {
EntryFile string
// Limits bounds files and sizes.
Limits Limits
// VerifySizes, when true, streams each regular file and compares the actual
// byte count against the archive-declared size (no content hashing).
// Default false: trust zip central directory / tar header sizes.
VerifySizes bool
}
// InspectFile opens path and inspects it as a Pages deployment package.
// InspectFile opens path and inspects it as a Pages deployment package without
// loading the whole archive into memory. File inventory uses declared sizes;
// per-file content hashes are not computed.
func InspectFile(filePath string, format Format, opts InspectOptions) (*Manifest, error) {
data, err := os.ReadFile(filePath) //nolint:gosec // filePath is a controlled temp upload path
file, err := os.Open(filePath) //nolint:gosec // filePath is a controlled temp upload path
if err != nil {
return nil, err
}
return InspectBytes(data, format, opts)
defer func() { _ = file.Close() }()
info, err := file.Stat()
if err != nil {
return nil, err
}
if format == "" {
head := make([]byte, 512)
n, readErr := file.ReadAt(head, 0)
if readErr != nil && readErr != io.EOF {
return nil, readErr
}
format, err = DetectFormat(filePath, head[:n])
if err != nil {
return nil, err
}
}
return inspectFromReaderAt(file, info.Size(), format, opts)
}
// InspectBytes inspects an in-memory deployment package.
@@ -38,7 +64,13 @@ func InspectBytes(data []byte, format Format, opts InspectOptions) (*Manifest, e
return nil, err
}
}
entries, err := listEntries(data, format)
return inspectFromReaderAt(bytes.NewReader(data), int64(len(data)), format, opts)
}
func inspectFromReaderAt(ra io.ReaderAt, size int64, format Format, opts InspectOptions) (*Manifest, error) {
// Default: zip/7z use central directory only; tar streams headers and discards bodies.
// VerifySizes needs openable tar bodies, so materialize only when requested.
entries, err := listEntriesAt(ra, size, format, opts.VerifySizes)
if err != nil {
return nil, err
}
@@ -65,7 +97,7 @@ func buildManifest(entries []Entry, opts InspectOptions) (*Manifest, error) {
return nil, fmt.Errorf("pages file too large: %s", normalizedPath)
}
fileEntry, err := inspectRegularFile(entry, normalizedPath, limits)
fileEntry, err := inspectRegularFile(entry, normalizedPath, limits, opts.VerifySizes)
if err != nil {
return nil, err
}
@@ -130,19 +162,44 @@ func prepareEntryPath(entry Entry, commonPrefix string) (string, bool, error) {
return normalizedPath, false, nil
}
func inspectRegularFile(entry Entry, normalizedPath string, limits Limits) (FileEntry, error) {
func inspectRegularFile(entry Entry, normalizedPath string, limits Limits, verifySizes bool) (FileEntry, error) {
if entry.Size > uint64(math.MaxInt64) {
return FileEntry{}, fmt.Errorf("%s: pages file size out of bounds", normalizedPath)
}
//nolint:gosec // bounded to MaxInt64 above
declaredSize := int64(entry.Size)
if !verifySizes {
return FileEntry{
Path: normalizedPath,
Size: declaredSize,
Checksum: "",
}, nil
}
if entry.Open == nil {
return FileEntry{}, fmt.Errorf("%s: cannot verify size without entry open", normalizedPath)
}
src, err := entry.Open()
if err != nil {
return FileEntry{}, fmt.Errorf("%s: %w", normalizedPath, err)
}
checksum, fileSize, checksumErr := checksumReader(src, entry.Size, limits.MaxFileBytes)
actual, measureErr := measureReader(src, entry.Size, limits.MaxFileBytes)
_ = src.Close()
if checksumErr != nil {
return FileEntry{}, fmt.Errorf("%s: %w", normalizedPath, checksumErr)
if measureErr != nil {
return FileEntry{}, fmt.Errorf("%s: %w", normalizedPath, measureErr)
}
if declaredSize > 0 && actual != declaredSize {
return FileEntry{}, fmt.Errorf("%s: declared size %d does not match actual %d", normalizedPath, declaredSize, actual)
}
return FileEntry{
Path: normalizedPath,
Size: fileSize,
Checksum: checksum,
Size: actual,
Checksum: "",
}, nil
}
// measureReader counts bytes without hashing, enforcing maxBytes when positive.
func measureReader(src io.Reader, declaredSize uint64, maxBytes int64) (int64, error) {
return copyLimited(io.Discard, src, declaredSize, maxBytes)
}
+80 -25
View File
@@ -53,31 +53,55 @@ func (z sevenZipArchiveFile) Open() (io.ReadCloser, error) {
return z.file.Open()
}
func listEntries(data []byte, format Format) ([]Entry, error) {
// listEntriesAt lists archive members from a random-access source.
// When materializeBodies is true, tar-family streams buffer regular-file bodies so Entry.Open works.
// When false (inspect path), tar bodies are discarded after reading headers; zip/7z only use central directory metadata.
func listEntriesAt(ra io.ReaderAt, size int64, format Format, materializeBodies bool) ([]Entry, error) {
if size < 0 {
return nil, fmt.Errorf("invalid pages package size")
}
switch format {
case FormatZip:
return listZipEntries(data)
return listZipEntriesAt(ra, size)
case FormatTar:
return listTarEntries(bytes.NewReader(data))
return listTarFamily(io.NewSectionReader(ra, 0, size), FormatTar, materializeBodies)
case FormatTarGz:
gzReader, err := gzip.NewReader(bytes.NewReader(data))
return listTarFamily(io.NewSectionReader(ra, 0, size), FormatTarGz, materializeBodies)
case FormatTarXz:
return listTarFamily(io.NewSectionReader(ra, 0, size), FormatTarXz, materializeBodies)
case FormatTarBz2:
return listTarFamily(io.NewSectionReader(ra, 0, size), FormatTarBz2, materializeBodies)
case FormatSevenZip:
return listSevenZipEntriesAt(ra, size)
default:
return nil, fmt.Errorf("unsupported pages package format: %s", format)
}
}
func listTarFamily(r io.Reader, format Format, materializeBodies bool) ([]Entry, error) {
switch format {
case FormatTar:
if materializeBodies {
return listTarEntries(r, true)
}
return listTarEntries(r, false)
case FormatTarGz:
gzReader, err := gzip.NewReader(r)
if err != nil {
return nil, fmt.Errorf("open gzip pages package: %w", err)
}
defer func() { _ = gzReader.Close() }()
return listTarEntries(gzReader)
return listTarEntries(gzReader, materializeBodies)
case FormatTarXz:
xzReader, err := xz.NewReader(bytes.NewReader(data))
xzReader, err := xz.NewReader(r)
if err != nil {
return nil, fmt.Errorf("open xz pages package: %w", err)
}
return listTarEntries(xzReader)
return listTarEntries(xzReader, materializeBodies)
case FormatTarBz2:
return listTarEntries(bzip2.NewReader(bytes.NewReader(data)))
case FormatSevenZip:
return listSevenZipEntries(data)
return listTarEntries(bzip2.NewReader(r), materializeBodies)
default:
return nil, fmt.Errorf("unsupported pages package format: %s", format)
return nil, fmt.Errorf("unsupported tar family format: %s", format)
}
}
@@ -96,8 +120,8 @@ func entriesFromArchiveFiles(files []archiveFile) []Entry {
return entries
}
func listZipEntries(data []byte) ([]Entry, error) {
reader, err := zip.NewReader(bytes.NewReader(data), int64(len(data)))
func listZipEntriesAt(ra io.ReaderAt, size int64) ([]Entry, error) {
reader, err := zip.NewReader(ra, size)
if err != nil {
return nil, fmt.Errorf("open zip pages package: %w", err)
}
@@ -108,8 +132,8 @@ func listZipEntries(data []byte) ([]Entry, error) {
return entriesFromArchiveFiles(files), nil
}
func listSevenZipEntries(data []byte) ([]Entry, error) {
reader, err := sevenzip.NewReader(bytes.NewReader(data), int64(len(data)))
func listSevenZipEntriesAt(ra io.ReaderAt, size int64) ([]Entry, error) {
reader, err := sevenzip.NewReader(ra, size)
if err != nil {
return nil, fmt.Errorf("open 7z pages package: %w", err)
}
@@ -120,9 +144,8 @@ func listSevenZipEntries(data []byte) ([]Entry, error) {
return entriesFromArchiveFiles(files), nil
}
func listTarEntries(r io.Reader) ([]Entry, error) {
func listTarEntries(r io.Reader, materializeBodies bool) ([]Entry, error) {
tarReader := tar.NewReader(r)
// Tar is sequential: materialize regular file bodies so entries can be opened later.
type materialised struct {
header *tar.Header
body []byte
@@ -136,7 +159,7 @@ func listTarEntries(r io.Reader) ([]Entry, error) {
if err != nil {
return nil, fmt.Errorf("read tar pages package: %w", err)
}
item, skip, err := materialiseTarHeader(tarReader, header)
item, skip, err := readTarHeader(tarReader, header, materializeBodies)
if err != nil {
return nil, err
}
@@ -148,12 +171,12 @@ func listTarEntries(r io.Reader) ([]Entry, error) {
entries := make([]Entry, 0, len(items))
for _, item := range items {
entries = append(entries, tarEntryFromMaterialised(item.header, item.body))
entries = append(entries, tarEntryFromHeader(item.header, item.body, materializeBodies))
}
return entries, nil
}
func materialiseTarHeader(tarReader *tar.Reader, header *tar.Header) (item struct {
func readTarHeader(tarReader *tar.Reader, header *tar.Header, materializeBodies bool) (item struct {
header *tar.Header
body []byte
}, skip bool, err error) {
@@ -164,6 +187,15 @@ func materialiseTarHeader(tarReader *tar.Reader, header *tar.Header) (item struc
body []byte
}{header: header}, false, nil
case tar.TypeReg, tar.TypeRegA: //nolint:staticcheck // TypeRegA still appears in older archives
if !materializeBodies {
if err := discardTarBody(tarReader, header); err != nil {
return item, false, err
}
return struct {
header *tar.Header
body []byte
}{header: header}, false, nil
}
body, readErr := readTarBody(tarReader, header)
if readErr != nil {
return item, false, readErr
@@ -182,6 +214,20 @@ func materialiseTarHeader(tarReader *tar.Reader, header *tar.Header) (item struc
}
}
func discardTarBody(tarReader *tar.Reader, header *tar.Header) error {
if header.Size <= 0 {
_, err := io.Copy(io.Discard, tarReader)
if err != nil {
return fmt.Errorf("discard tar entry %s: %w", header.Name, err)
}
return nil
}
if _, err := io.CopyN(io.Discard, tarReader, header.Size); err != nil {
return fmt.Errorf("discard tar entry %s: %w", header.Name, err)
}
return nil
}
func readTarBody(tarReader *tar.Reader, header *tar.Header) ([]byte, error) {
if header.Size > 0 {
body := make([]byte, header.Size)
@@ -197,9 +243,9 @@ func readTarBody(tarReader *tar.Reader, header *tar.Header) ([]byte, error) {
return body, nil
}
func tarEntryFromMaterialised(header *tar.Header, body []byte) Entry {
func tarEntryFromHeader(header *tar.Header, body []byte, materializeBodies bool) Entry {
size := header.Size
if int64(len(body)) > size {
if materializeBodies && int64(len(body)) > size {
size = int64(len(body))
}
entry := Entry{
@@ -207,14 +253,23 @@ func tarEntryFromMaterialised(header *tar.Header, body []byte) Entry {
IsDir: header.Typeflag == tar.TypeDir,
IsSymlink: header.Typeflag == tar.TypeSymlink || header.Typeflag == tar.TypeLink,
Size: uint64(size), //nolint:gosec // non-negative sizes
Open: func() (io.ReadCloser, error) {
return io.NopCloser(bytes.NewReader(body)), nil
},
}
if entry.IsDir || entry.IsSymlink {
entry.Open = func() (io.ReadCloser, error) {
return io.NopCloser(bytes.NewReader(nil)), nil
}
return entry
}
if materializeBodies {
bodyCopy := body
entry.Open = func() (io.ReadCloser, error) {
return io.NopCloser(bytes.NewReader(bodyCopy)), nil
}
return entry
}
// Inspect path: body not retained; Open is unavailable.
entry.Open = func() (io.ReadCloser, error) {
return nil, fmt.Errorf("tar entry body not materialized: %s", header.Name)
}
return entry
}
+46
View File
@@ -52,9 +52,12 @@ func TestInspectAndExtractZip(t *testing.T) {
paths := make(map[string]struct{}, len(manifest.Files))
for _, file := range manifest.Files {
paths[file.Path] = struct{}{}
assert.Empty(t, file.Checksum, "per-file checksum should not be computed")
assert.Positive(t, file.Size)
}
assert.Contains(t, paths, "index.html")
assert.Contains(t, paths, "app.js")
assert.Equal(t, int64(len("<html>ok</html>")+len("console.log(1)")), manifest.TotalSize)
dest := t.TempDir()
require.NoError(t, ExtractBytes(data, FormatZip, dest, ExtractOptions{
@@ -67,6 +70,49 @@ func TestInspectAndExtractZip(t *testing.T) {
assert.Equal(t, "<html>ok</html>", string(body))
}
func TestInspectFileUsesDeclaredSizesWithoutHash(t *testing.T) {
data := testZip(t, map[string]string{
"index.html": "<html>disk</html>",
"asset.css": "body{}",
})
path := filepath.Join(t.TempDir(), "site.zip")
require.NoError(t, os.WriteFile(path, data, 0o600))
manifest, err := InspectFile(path, FormatZip, InspectOptions{
EntryFile: "index.html",
Limits: Limits{MaxFiles: 100, MaxFileBytes: 1 << 20, MaxTotalBytes: 1 << 20},
})
require.NoError(t, err)
require.Equal(t, 2, manifest.FileCount)
for _, file := range manifest.Files {
assert.Empty(t, file.Checksum)
}
dest := t.TempDir()
require.NoError(t, ExtractFile(path, FormatZip, dest, ExtractOptions{
EnforceLimits: true,
Limits: Limits{MaxFiles: 100, MaxFileBytes: 1 << 20, MaxTotalBytes: 1 << 20},
}))
body, err := os.ReadFile(filepath.Join(dest, "index.html")) //nolint:gosec
require.NoError(t, err)
assert.Equal(t, "<html>disk</html>", string(body))
}
func TestInspectVerifySizesOptional(t *testing.T) {
data := testZip(t, map[string]string{
"index.html": "verify-me",
})
manifest, err := InspectBytes(data, FormatZip, InspectOptions{
EntryFile: "index.html",
VerifySizes: true,
Limits: Limits{MaxFiles: 10, MaxFileBytes: 1 << 20, MaxTotalBytes: 1 << 20},
})
require.NoError(t, err)
require.Len(t, manifest.Files, 1)
assert.Equal(t, int64(len("verify-me")), manifest.Files[0].Size)
assert.Empty(t, manifest.Files[0].Checksum)
}
func TestExtractTrustedSkipsSizeLimits(t *testing.T) {
// Content larger than a tiny limit would fail if limits were enforced.
large := strings.Repeat("x", 64)