feat(pages): configurable limits, multi-format packages, and dual versioning

Make Pages package size and history retention system-configurable, support
zip/tar.gz/tar.xz/tar.bz2/tar/7z uploads, prune history with clear keep-N
semantics, and rebind agent config to the live active Pages deployment so
main-config rollback never depends on pruned packages.
This commit is contained in:
ryan
2026-07-17 17:16:48 +08:00
parent 368df3f76b
commit a0fcf9f627
32 changed files with 2286 additions and 370 deletions
+101
View File
@@ -0,0 +1,101 @@
// Copyright 2026 Arctel.net
// SPDX-License-Identifier: Apache-2.0
package pagesarchive
import (
"crypto/sha256"
"encoding/hex"
"fmt"
"io"
"math"
"os"
"path/filepath"
)
// Limits bounds archive inspection / extraction work.
type Limits struct {
// MaxFiles is the maximum number of regular files allowed.
MaxFiles int
// MaxFileBytes is the maximum size of a single extracted file.
MaxFileBytes int64
// MaxTotalBytes is the maximum sum of all extracted file sizes.
MaxTotalBytes int64
}
// FileEntry is a regular file discovered inside a deployment package.
type FileEntry struct {
Path string
Size int64
Checksum string
}
// Manifest is the inspected content of a Pages deployment package.
type Manifest struct {
Files []FileEntry
FileCount int
TotalSize int64
}
// Entry describes one archive member for extraction.
type Entry struct {
// Name is the original path inside the archive.
Name string
// IsDir marks directory entries.
IsDir bool
// IsSymlink marks symbolic links (unsupported for Pages).
IsSymlink bool
// Size is the declared uncompressed size when known; 0 means unknown.
Size uint64
// Open returns a reader for the entry body. Caller must Close it.
Open func() (io.ReadCloser, error)
}
// copyLimited copies src to dst.
// When maxBytes <= 0, size limits are not enforced (trusted extract path).
func copyLimited(dst io.Writer, src io.Reader, declaredSize uint64, maxBytes int64) (int64, error) {
if maxBytes <= 0 {
if declaredSize > 0 {
if declaredSize > uint64(math.MaxInt64) {
return 0, fmt.Errorf("pages file size out of bounds")
}
//nolint:gosec // declaredSize is bounded to MaxInt64 above
return io.CopyN(dst, src, int64(declaredSize))
}
return io.Copy(dst, src)
}
if declaredSize > uint64(maxBytes) || declaredSize > uint64(math.MaxInt64) { //nolint:gosec // maxBytes positive
return 0, fmt.Errorf("pages file size out of bounds")
}
if declaredSize > 0 {
//nolint:gosec // declaredSize is bounded to MaxInt64 above
return io.CopyN(dst, src, int64(declaredSize))
}
limited := io.LimitReader(src, maxBytes+1)
written, err := io.Copy(dst, limited)
if written > maxBytes {
return written, fmt.Errorf("pages file size out of bounds")
}
return written, err
}
func checksumReader(src io.Reader, declaredSize uint64, maxBytes int64) (string, int64, error) {
hash := sha256.New()
written, err := copyLimited(hash, src, declaredSize, maxBytes)
if err != nil {
return "", written, err
}
return hex.EncodeToString(hash.Sum(nil)), written, nil
}
func writeEntryFile(targetPath string, src io.Reader, declaredSize uint64, maxBytes int64, perm os.FileMode) (int64, error) {
if err := os.MkdirAll(filepath.Dir(targetPath), dirPerm); err != nil {
return 0, err
}
target, err := os.OpenFile(targetPath, os.O_CREATE|os.O_WRONLY|os.O_TRUNC, perm) //nolint:gosec // caller validates path under release dir
if err != nil {
return 0, err
}
defer func() { _ = target.Close() }()
return copyLimited(target, src, declaredSize, maxBytes)
}
+154
View File
@@ -0,0 +1,154 @@
// Copyright 2026 Arctel.net
// SPDX-License-Identifier: Apache-2.0
package pagesarchive
import (
"fmt"
"os"
"path/filepath"
)
// ExtractOptions controls package extraction.
type ExtractOptions struct {
// Limits bounds files and sizes during extraction when EnforceLimits is true.
Limits Limits
// StripCommonRoot strips a single shared top-level directory when present.
StripCommonRoot bool
// EnforceLimits enables MaxFiles / MaxFileBytes / MaxTotalBytes checks.
// When false, the caller is assumed to have already validated the package
// (e.g. Agent trusts control-plane inspection). Path-escape and symlink
// guards still apply so local extraction cannot leave destDir.
EnforceLimits bool
}
// ExtractBytes extracts a deployment package into destDir.
func ExtractBytes(data []byte, format Format, destDir string, opts ExtractOptions) error {
if format == "" {
var err error
format, err = DetectFormat("", data)
if err != nil {
return err
}
}
entries, err := listEntries(data, format)
if err != nil {
return err
}
return extractEntries(entries, destDir, opts)
}
// ExtractFile reads path and extracts it into destDir.
func ExtractFile(filePath string, format Format, destDir string, opts ExtractOptions) error {
data, err := os.ReadFile(filePath) //nolint:gosec // controlled path
if err != nil {
return err
}
return ExtractBytes(data, format, destDir, opts)
}
func extractEntries(entries []Entry, destDir string, opts ExtractOptions) error {
limits := Limits{}
if opts.EnforceLimits {
limits = normalizeLimits(opts.Limits)
}
commonPrefix := ""
if opts.StripCommonRoot {
commonPrefix = FindCommonRootPrefix(collectFileNames(entries))
}
var totalSize int64
var fileCount int
for _, entry := range entries {
written, counted, err := extractSingleEntry(entry, destDir, commonPrefix, limits, opts.EnforceLimits)
if err != nil {
return err
}
if !counted {
continue
}
fileCount++
if opts.EnforceLimits && fileCount > limits.MaxFiles {
return fmt.Errorf("pages deployment file count exceeds %d", limits.MaxFiles)
}
totalSize += written
if opts.EnforceLimits && totalSize > limits.MaxTotalBytes {
return fmt.Errorf("pages extracted size exceeds limit")
}
}
if fileCount == 0 {
return fmt.Errorf("pages package is empty")
}
return nil
}
func extractSingleEntry(
entry Entry,
destDir, commonPrefix string,
limits Limits,
enforceLimits bool,
) (written int64, counted bool, err error) {
relativePath, skip, err := NormalizeEntryPath(entry.Name)
if err != nil {
return 0, false, err
}
if skip {
return 0, false, nil
}
if commonPrefix != "" {
relativePath = StripPrefix(relativePath, commonPrefix)
if relativePath == "" {
return 0, false, nil
}
}
if entry.IsSymlink {
return 0, false, fmt.Errorf("pages package contains unsupported symlink: %s", relativePath)
}
targetPath := filepath.Join(destDir, filepath.FromSlash(relativePath))
if !isWithinDir(destDir, targetPath) {
return 0, false, fmt.Errorf("pages package path escapes directory: %s", entry.Name)
}
if entry.IsDir {
if err := os.MkdirAll(targetPath, dirPerm); err != nil {
return 0, false, err
}
return 0, false, nil
}
maxFileBytes := int64(0) // unlimited when not enforcing
if enforceLimits {
if exceedsFileByteLimit(entry.Size, limits.MaxFileBytes) {
return 0, false, fmt.Errorf("pages file too large: %s", relativePath)
}
maxFileBytes = limits.MaxFileBytes
}
src, err := entry.Open()
if err != nil {
return 0, false, fmt.Errorf("%s: %w", relativePath, err)
}
written, writeErr := writeEntryFile(targetPath, src, entry.Size, maxFileBytes, filePerm)
_ = src.Close()
if writeErr != nil {
return 0, false, fmt.Errorf("%s: %w", relativePath, writeErr)
}
return written, true, nil
}
func isWithinDir(baseDir, targetPath string) bool {
cleanBase := filepath.Clean(baseDir)
cleanTarget := filepath.Clean(targetPath)
rel, err := filepath.Rel(cleanBase, cleanTarget)
if err != nil {
return false
}
return rel != ".." && !hasParentRel(rel)
}
func hasParentRel(rel string) bool {
if rel == ".." {
return true
}
return len(rel) >= 3 && (rel[:3] == "../" || rel[:3] == "..\\")
}
+191
View File
@@ -0,0 +1,191 @@
// Copyright 2026 Arctel.net
// SPDX-License-Identifier: Apache-2.0
// Package pagesarchive provides multi-format archive detection, inspection and
// extraction helpers for OpenFlare Pages deployment packages.
package pagesarchive
import (
"bytes"
"fmt"
"path/filepath"
"strings"
)
// Format identifies a supported Pages deployment package archive format.
type Format string
const (
// FormatZip is a ZIP archive.
FormatZip Format = "zip"
// FormatTar is an uncompressed tar archive.
FormatTar Format = "tar"
// FormatTarGz is a gzip-compressed tar archive.
FormatTarGz Format = "tar.gz"
// FormatTarXz is an xz-compressed tar archive.
FormatTarXz Format = "tar.xz"
// FormatTarBz2 is a bzip2-compressed tar archive.
FormatTarBz2 Format = "tar.bz2"
// FormatSevenZip is a 7z archive.
FormatSevenZip Format = "7z"
ustarMagicOffset = 257
ustarMagicMinLen = 262
)
// DetectFormatFromName returns the archive format inferred from a file name.
// Returns an empty Format and false when the extension is unsupported.
func DetectFormatFromName(fileName string) (Format, bool) {
name := strings.ToLower(strings.TrimSpace(fileName))
switch {
case strings.HasSuffix(name, ".tar.gz"), strings.HasSuffix(name, ".tgz"):
return FormatTarGz, true
case strings.HasSuffix(name, ".tar.xz"), strings.HasSuffix(name, ".txz"):
return FormatTarXz, true
case strings.HasSuffix(name, ".tar.bz2"), strings.HasSuffix(name, ".tbz2"), strings.HasSuffix(name, ".tbz"):
return FormatTarBz2, true
case strings.HasSuffix(name, ".tar"):
return FormatTar, true
case strings.HasSuffix(name, ".7z"):
return FormatSevenZip, true
case strings.HasSuffix(name, ".zip"):
return FormatZip, true
default:
return "", false
}
}
// DetectFormatFromBytes returns the archive format inferred from magic bytes.
// Prefer DetectFormatFromName when a reliable file name is available.
func DetectFormatFromBytes(data []byte) (Format, bool) {
if isZipMagic(data) {
return FormatZip, true
}
if isSevenZipMagic(data) {
return FormatSevenZip, true
}
if isXZMagic(data) {
return FormatTarXz, true
}
if isGzipMagic(data) {
return FormatTarGz, true
}
if isBzip2Magic(data) {
return FormatTarBz2, true
}
if looksLikeTar(data) {
return FormatTar, true
}
return "", false
}
// DetectFormat prefers the file name when present, otherwise magic bytes.
func DetectFormat(fileName string, data []byte) (Format, error) {
if format, ok := DetectFormatFromName(fileName); ok {
return format, nil
}
if format, ok := DetectFormatFromBytes(data); ok {
return format, nil
}
return "", fmt.Errorf("unsupported pages package format")
}
// Extension returns the canonical file extension for a format (without leading dot).
func Extension(format Format) string {
switch format {
case FormatZip:
return "zip"
case FormatTar:
return "tar"
case FormatTarGz:
return "tar.gz"
case FormatTarXz:
return "tar.xz"
case FormatTarBz2:
return "tar.bz2"
case FormatSevenZip:
return "7z"
default:
return "bin"
}
}
// MIMEType returns a reasonable content type for the archive format.
func MIMEType(format Format) string {
switch format {
case FormatZip:
return "application/zip"
case FormatTar:
return "application/x-tar"
case FormatTarGz:
return "application/gzip"
case FormatTarXz:
return "application/x-xz"
case FormatTarBz2:
return "application/x-bzip2"
case FormatSevenZip:
return "application/x-7z-compressed"
default:
return "application/octet-stream"
}
}
// SupportedExtensions lists human-readable extensions for UI copy and accept attributes.
func SupportedExtensions() []string {
return []string{".zip", ".tar.gz", ".tgz", ".tar.xz", ".txz", ".tar.bz2", ".tbz2", ".tar", ".7z"}
}
// AcceptAttribute returns a comma-separated accept list for file inputs.
func AcceptAttribute() string {
return strings.Join(SupportedExtensions(), ",")
}
// NormalizeNameExtension returns a storage-safe extension for the given format/name.
func NormalizeNameExtension(fileName string, format Format) string {
if format != "" {
return Extension(format)
}
if formatFromName, ok := DetectFormatFromName(fileName); ok {
return Extension(formatFromName)
}
ext := strings.TrimPrefix(filepath.Ext(fileName), ".")
if ext == "" {
return "bin"
}
return strings.ToLower(ext)
}
func isZipMagic(data []byte) bool {
return len(data) >= 4 &&
data[0] == 0x50 && data[1] == 0x4b &&
(data[2] == 0x03 || data[2] == 0x05 || data[2] == 0x07) &&
(data[3] == 0x04 || data[3] == 0x06 || data[3] == 0x08)
}
func isSevenZipMagic(data []byte) bool {
return len(data) >= 6 &&
data[0] == 0x37 && data[1] == 0x7a && data[2] == 0xbc &&
data[3] == 0xaf && data[4] == 0x27 && data[5] == 0x1c
}
func isXZMagic(data []byte) bool {
return len(data) >= 6 &&
data[0] == 0xfd && data[1] == 0x37 && data[2] == 0x7a &&
data[3] == 0x58 && data[4] == 0x5a && data[5] == 0x00
}
func isGzipMagic(data []byte) bool {
return len(data) >= 2 && data[0] == 0x1f && data[1] == 0x8b
}
func isBzip2Magic(data []byte) bool {
return len(data) >= 3 && data[0] == 0x42 && data[1] == 0x5a && data[2] == 0x68
}
func looksLikeTar(data []byte) bool {
// POSIX ustar magic at offset 257 ("ustar\0" or "ustar ").
if len(data) < ustarMagicMinLen {
return false
}
return bytes.Equal(data[ustarMagicOffset:ustarMagicOffset+5], []byte("ustar"))
}
+148
View File
@@ -0,0 +1,148 @@
// Copyright 2026 Arctel.net
// SPDX-License-Identifier: Apache-2.0
package pagesarchive
import (
"fmt"
"os"
"path"
"strings"
)
// InspectOptions controls package inspection.
type InspectOptions struct {
// RootDir is an optional project root subdirectory that must contain EntryFile.
RootDir string
// EntryFile is the required entry file name (e.g. index.html).
EntryFile string
// Limits bounds files and sizes.
Limits Limits
}
// InspectFile opens path and inspects it as a Pages deployment package.
func InspectFile(filePath string, format Format, opts InspectOptions) (*Manifest, error) {
data, err := os.ReadFile(filePath) //nolint:gosec // filePath is a controlled temp upload path
if err != nil {
return nil, err
}
return InspectBytes(data, format, opts)
}
// InspectBytes inspects an in-memory deployment package.
func InspectBytes(data []byte, format Format, opts InspectOptions) (*Manifest, error) {
if format == "" {
var err error
format, err = DetectFormat("", data)
if err != nil {
return nil, err
}
}
entries, err := listEntries(data, format)
if err != nil {
return nil, err
}
return buildManifest(entries, opts)
}
func buildManifest(entries []Entry, opts InspectOptions) (*Manifest, error) {
limits := normalizeLimits(opts.Limits)
commonPrefix := FindCommonRootPrefix(collectFileNames(entries))
targetEntryPath := resolveTargetEntryPath(opts.RootDir, opts.EntryFile)
manifest := &Manifest{Files: make([]FileEntry, 0)}
entrySeen := false
for _, entry := range entries {
normalizedPath, skip, err := prepareEntryPath(entry, commonPrefix)
if err != nil {
return nil, err
}
if skip {
continue
}
if exceedsFileByteLimit(entry.Size, limits.MaxFileBytes) {
return nil, fmt.Errorf("pages file too large: %s", normalizedPath)
}
fileEntry, err := inspectRegularFile(entry, normalizedPath, limits)
if err != nil {
return nil, err
}
manifest.FileCount++
if manifest.FileCount > limits.MaxFiles {
return nil, fmt.Errorf("pages deployment file count exceeds %d", limits.MaxFiles)
}
manifest.TotalSize += fileEntry.Size
if manifest.TotalSize > limits.MaxTotalBytes {
return nil, fmt.Errorf("pages extracted size exceeds limit")
}
if normalizedPath == targetEntryPath {
entrySeen = true
}
manifest.Files = append(manifest.Files, fileEntry)
}
if manifest.FileCount == 0 {
return nil, fmt.Errorf("pages package is empty")
}
if !entrySeen {
return nil, fmt.Errorf("pages package is missing entry file %s", targetEntryPath)
}
return manifest, nil
}
func collectFileNames(entries []Entry) []string {
names := make([]string, 0, len(entries))
for _, entry := range entries {
if entry.IsDir || entry.IsSymlink {
continue
}
names = append(names, entry.Name)
}
return names
}
func resolveTargetEntryPath(rootDir, entryFile string) string {
normalizedEntry := strings.TrimSpace(entryFile)
if normalizedEntry == "" {
normalizedEntry = "index.html"
}
normalizedRoot := strings.Trim(strings.TrimSpace(rootDir), "/")
if normalizedRoot == "" {
return normalizedEntry
}
return path.Join(normalizedRoot, normalizedEntry)
}
func prepareEntryPath(entry Entry, commonPrefix string) (string, bool, error) {
normalizedPath, skip, err := NormalizeEntryPath(entry.Name)
if err != nil {
return "", false, err
}
if skip || entry.IsDir {
return "", true, nil
}
normalizedPath = StripPrefix(normalizedPath, commonPrefix)
if entry.IsSymlink {
return "", false, fmt.Errorf("pages package contains unsupported symlink: %s", normalizedPath)
}
return normalizedPath, false, nil
}
func inspectRegularFile(entry Entry, normalizedPath string, limits Limits) (FileEntry, error) {
src, err := entry.Open()
if err != nil {
return FileEntry{}, fmt.Errorf("%s: %w", normalizedPath, err)
}
checksum, fileSize, checksumErr := checksumReader(src, entry.Size, limits.MaxFileBytes)
_ = src.Close()
if checksumErr != nil {
return FileEntry{}, fmt.Errorf("%s: %w", normalizedPath, checksumErr)
}
return FileEntry{
Path: normalizedPath,
Size: fileSize,
Checksum: checksum,
}, nil
}
+38
View File
@@ -0,0 +1,38 @@
// Copyright 2026 Arctel.net
// SPDX-License-Identifier: Apache-2.0
package pagesarchive
const (
defaultMaxFiles = 1000
defaultMaxFileBytes = 100 * 1024 * 1024
defaultMaxTotalBytes = 100 * 1024 * 1024
dirPerm = 0o750
filePerm = 0o644
)
// normalizeLimits applies defaults for control-plane inspection.
// Callers that already validated the package should use EnforceLimits=false instead.
func normalizeLimits(limits Limits) Limits {
if limits.MaxFiles <= 0 {
limits.MaxFiles = defaultMaxFiles
}
if limits.MaxFileBytes <= 0 {
limits.MaxFileBytes = defaultMaxFileBytes
}
if limits.MaxTotalBytes <= 0 {
limits.MaxTotalBytes = defaultMaxTotalBytes
}
return limits
}
func exceedsFileByteLimit(size uint64, maxBytes int64) bool {
// maxBytes <= 0 means unlimited (trusted extract path).
if maxBytes <= 0 {
return false
}
if size == 0 {
return false
}
return size > uint64(maxBytes) //nolint:gosec // maxBytes is positive
}
+220
View File
@@ -0,0 +1,220 @@
// Copyright 2026 Arctel.net
// SPDX-License-Identifier: Apache-2.0
package pagesarchive
import (
"archive/tar"
"archive/zip"
"bytes"
"compress/bzip2"
"compress/gzip"
"fmt"
"io"
"os"
"github.com/bodgit/sevenzip"
"github.com/ulikunitz/xz"
)
type archiveFile interface {
Name() string
IsDir() bool
IsSymlink() bool
Size() uint64
Open() (io.ReadCloser, error)
}
type zipArchiveFile struct {
file *zip.File
}
func (z zipArchiveFile) Name() string { return z.file.Name }
func (z zipArchiveFile) IsDir() bool { return z.file.FileInfo().IsDir() }
func (z zipArchiveFile) IsSymlink() bool {
return z.file.Mode()&os.ModeSymlink != 0
}
func (z zipArchiveFile) Size() uint64 { return z.file.UncompressedSize64 }
func (z zipArchiveFile) Open() (io.ReadCloser, error) {
return z.file.Open()
}
type sevenZipArchiveFile struct {
file *sevenzip.File
}
func (z sevenZipArchiveFile) Name() string { return z.file.Name }
func (z sevenZipArchiveFile) IsDir() bool { return z.file.FileInfo().IsDir() }
func (z sevenZipArchiveFile) IsSymlink() bool {
return z.file.Mode()&os.ModeSymlink != 0
}
func (z sevenZipArchiveFile) Size() uint64 { return z.file.UncompressedSize }
func (z sevenZipArchiveFile) Open() (io.ReadCloser, error) {
return z.file.Open()
}
func listEntries(data []byte, format Format) ([]Entry, error) {
switch format {
case FormatZip:
return listZipEntries(data)
case FormatTar:
return listTarEntries(bytes.NewReader(data))
case FormatTarGz:
gzReader, err := gzip.NewReader(bytes.NewReader(data))
if err != nil {
return nil, fmt.Errorf("open gzip pages package: %w", err)
}
defer func() { _ = gzReader.Close() }()
return listTarEntries(gzReader)
case FormatTarXz:
xzReader, err := xz.NewReader(bytes.NewReader(data))
if err != nil {
return nil, fmt.Errorf("open xz pages package: %w", err)
}
return listTarEntries(xzReader)
case FormatTarBz2:
return listTarEntries(bzip2.NewReader(bytes.NewReader(data)))
case FormatSevenZip:
return listSevenZipEntries(data)
default:
return nil, fmt.Errorf("unsupported pages package format: %s", format)
}
}
func entriesFromArchiveFiles(files []archiveFile) []Entry {
entries := make([]Entry, 0, len(files))
for _, item := range files {
file := item
entries = append(entries, Entry{
Name: file.Name(),
IsDir: file.IsDir(),
IsSymlink: file.IsSymlink(),
Size: file.Size(),
Open: file.Open,
})
}
return entries
}
func listZipEntries(data []byte) ([]Entry, error) {
reader, err := zip.NewReader(bytes.NewReader(data), int64(len(data)))
if err != nil {
return nil, fmt.Errorf("open zip pages package: %w", err)
}
files := make([]archiveFile, 0, len(reader.File))
for _, item := range reader.File {
files = append(files, zipArchiveFile{file: item})
}
return entriesFromArchiveFiles(files), nil
}
func listSevenZipEntries(data []byte) ([]Entry, error) {
reader, err := sevenzip.NewReader(bytes.NewReader(data), int64(len(data)))
if err != nil {
return nil, fmt.Errorf("open 7z pages package: %w", err)
}
files := make([]archiveFile, 0, len(reader.File))
for _, item := range reader.File {
files = append(files, sevenZipArchiveFile{file: item})
}
return entriesFromArchiveFiles(files), nil
}
func listTarEntries(r io.Reader) ([]Entry, error) {
tarReader := tar.NewReader(r)
// Tar is sequential: materialize regular file bodies so entries can be opened later.
type materialised struct {
header *tar.Header
body []byte
}
items := make([]materialised, 0)
for {
header, err := tarReader.Next()
if err == io.EOF {
break
}
if err != nil {
return nil, fmt.Errorf("read tar pages package: %w", err)
}
item, skip, err := materialiseTarHeader(tarReader, header)
if err != nil {
return nil, err
}
if skip {
continue
}
items = append(items, item)
}
entries := make([]Entry, 0, len(items))
for _, item := range items {
entries = append(entries, tarEntryFromMaterialised(item.header, item.body))
}
return entries, nil
}
func materialiseTarHeader(tarReader *tar.Reader, header *tar.Header) (item struct {
header *tar.Header
body []byte
}, skip bool, err error) {
switch header.Typeflag {
case tar.TypeDir, tar.TypeSymlink, tar.TypeLink:
return struct {
header *tar.Header
body []byte
}{header: header}, false, nil
case tar.TypeReg, tar.TypeRegA: //nolint:staticcheck // TypeRegA still appears in older archives
body, readErr := readTarBody(tarReader, header)
if readErr != nil {
return item, false, readErr
}
return struct {
header *tar.Header
body []byte
}{header: header, body: body}, false, nil
default:
if header.Size > 0 {
if _, copyErr := io.CopyN(io.Discard, tarReader, header.Size); copyErr != nil {
return item, false, fmt.Errorf("skip tar entry %s: %w", header.Name, copyErr)
}
}
return item, true, nil
}
}
func readTarBody(tarReader *tar.Reader, header *tar.Header) ([]byte, error) {
if header.Size > 0 {
body := make([]byte, header.Size)
if _, err := io.ReadFull(tarReader, body); err != nil {
return nil, fmt.Errorf("read tar entry %s: %w", header.Name, err)
}
return body, nil
}
body, err := io.ReadAll(tarReader)
if err != nil {
return nil, fmt.Errorf("read tar entry %s: %w", header.Name, err)
}
return body, nil
}
func tarEntryFromMaterialised(header *tar.Header, body []byte) Entry {
size := header.Size
if int64(len(body)) > size {
size = int64(len(body))
}
entry := Entry{
Name: header.Name,
IsDir: header.Typeflag == tar.TypeDir,
IsSymlink: header.Typeflag == tar.TypeSymlink || header.Typeflag == tar.TypeLink,
Size: uint64(size), //nolint:gosec // non-negative sizes
Open: func() (io.ReadCloser, error) {
return io.NopCloser(bytes.NewReader(body)), nil
},
}
if entry.IsDir || entry.IsSymlink {
entry.Open = func() (io.ReadCloser, error) {
return io.NopCloser(bytes.NewReader(nil)), nil
}
}
return entry
}
+187
View File
@@ -0,0 +1,187 @@
// Copyright 2026 Arctel.net
// SPDX-License-Identifier: Apache-2.0
package pagesarchive
import (
"archive/tar"
"archive/zip"
"bytes"
"compress/gzip"
"os"
"path/filepath"
"strings"
"testing"
"github.com/stretchr/testify/assert"
"github.com/stretchr/testify/require"
"github.com/ulikunitz/xz"
)
func TestDetectFormatFromName(t *testing.T) {
cases := map[string]Format{
"site.zip": FormatZip,
"site.TAR.GZ": FormatTarGz,
"site.tgz": FormatTarGz,
"site.tar.xz": FormatTarXz,
"site.txz": FormatTarXz,
"site.tar.bz2": FormatTarBz2,
"site.tar": FormatTar,
"site.7z": FormatSevenZip,
}
for name, want := range cases {
got, ok := DetectFormatFromName(name)
assert.True(t, ok, name)
assert.Equal(t, want, got, name)
}
_, ok := DetectFormatFromName("site.rar")
assert.False(t, ok)
}
func TestInspectAndExtractZip(t *testing.T) {
data := testZip(t, map[string]string{
"dist/index.html": "<html>ok</html>",
"dist/app.js": "console.log(1)",
})
manifest, err := InspectBytes(data, FormatZip, InspectOptions{
EntryFile: "index.html",
Limits: Limits{MaxFiles: 100, MaxFileBytes: 1 << 20, MaxTotalBytes: 1 << 20},
})
require.NoError(t, err)
assert.Equal(t, 2, manifest.FileCount)
paths := make(map[string]struct{}, len(manifest.Files))
for _, file := range manifest.Files {
paths[file.Path] = struct{}{}
}
assert.Contains(t, paths, "index.html")
assert.Contains(t, paths, "app.js")
dest := t.TempDir()
require.NoError(t, ExtractBytes(data, FormatZip, dest, ExtractOptions{
StripCommonRoot: true,
EnforceLimits: true,
Limits: Limits{MaxFiles: 100, MaxFileBytes: 1 << 20, MaxTotalBytes: 1 << 20},
}))
body, err := os.ReadFile(filepath.Join(dest, "index.html")) //nolint:gosec
require.NoError(t, err)
assert.Equal(t, "<html>ok</html>", string(body))
}
func TestExtractTrustedSkipsSizeLimits(t *testing.T) {
// Content larger than a tiny limit would fail if limits were enforced.
large := strings.Repeat("x", 64)
data := testZip(t, map[string]string{
"index.html": large,
})
dest := t.TempDir()
require.NoError(t, ExtractBytes(data, FormatZip, dest, ExtractOptions{
// Agent trusts control-plane validation: no size/count re-check.
EnforceLimits: false,
}))
body, err := os.ReadFile(filepath.Join(dest, "index.html")) //nolint:gosec
require.NoError(t, err)
assert.Equal(t, large, string(body))
}
func TestInspectAndExtractTarGz(t *testing.T) {
data := testTarGz(t, map[string]string{
"index.html": "<html>tar</html>",
"style.css": "body{}",
})
format, err := DetectFormat("site.tar.gz", data)
require.NoError(t, err)
assert.Equal(t, FormatTarGz, format)
manifest, err := InspectBytes(data, format, InspectOptions{
EntryFile: "index.html",
Limits: Limits{MaxFiles: 100, MaxFileBytes: 1 << 20, MaxTotalBytes: 1 << 20},
})
require.NoError(t, err)
assert.Equal(t, 2, manifest.FileCount)
dest := t.TempDir()
require.NoError(t, ExtractBytes(data, format, dest, ExtractOptions{
EnforceLimits: true,
Limits: Limits{MaxFiles: 100, MaxFileBytes: 1 << 20, MaxTotalBytes: 1 << 20},
}))
body, err := os.ReadFile(filepath.Join(dest, "index.html")) //nolint:gosec
require.NoError(t, err)
assert.Equal(t, "<html>tar</html>", string(body))
}
func TestInspectTarXz(t *testing.T) {
data := testTarXz(t, map[string]string{
"index.html": "<html>xz</html>",
})
manifest, err := InspectBytes(data, FormatTarXz, InspectOptions{
EntryFile: "index.html",
Limits: Limits{MaxFiles: 10, MaxFileBytes: 1 << 20, MaxTotalBytes: 1 << 20},
})
require.NoError(t, err)
assert.Equal(t, 1, manifest.FileCount)
}
func TestRejectZipSlip(t *testing.T) {
data := testZip(t, map[string]string{
"../evil.txt": "x",
"index.html": "ok",
})
_, err := InspectBytes(data, FormatZip, InspectOptions{
EntryFile: "index.html",
Limits: Limits{MaxFiles: 10, MaxFileBytes: 1 << 20, MaxTotalBytes: 1 << 20},
})
require.Error(t, err)
}
func testZip(t *testing.T, files map[string]string) []byte {
t.Helper()
var buffer bytes.Buffer
writer := zip.NewWriter(&buffer)
for name, content := range files {
file, err := writer.Create(name)
require.NoError(t, err)
_, err = file.Write([]byte(content))
require.NoError(t, err)
}
require.NoError(t, writer.Close())
return buffer.Bytes()
}
func testTarGz(t *testing.T, files map[string]string) []byte {
t.Helper()
var buffer bytes.Buffer
gzWriter := gzip.NewWriter(&buffer)
tarWriter := tar.NewWriter(gzWriter)
for name, content := range files {
require.NoError(t, tarWriter.WriteHeader(&tar.Header{
Name: name,
Mode: 0o644,
Size: int64(len(content)),
}))
_, err := tarWriter.Write([]byte(content))
require.NoError(t, err)
}
require.NoError(t, tarWriter.Close())
require.NoError(t, gzWriter.Close())
return buffer.Bytes()
}
func testTarXz(t *testing.T, files map[string]string) []byte {
t.Helper()
var buffer bytes.Buffer
xzWriter, err := xz.NewWriter(&buffer)
require.NoError(t, err)
tarWriter := tar.NewWriter(xzWriter)
for name, content := range files {
require.NoError(t, tarWriter.WriteHeader(&tar.Header{
Name: name,
Mode: 0o644,
Size: int64(len(content)),
}))
_, err := tarWriter.Write([]byte(content))
require.NoError(t, err)
}
require.NoError(t, tarWriter.Close())
require.NoError(t, xzWriter.Close())
return buffer.Bytes()
}
+85
View File
@@ -0,0 +1,85 @@
// Copyright 2026 Arctel.net
// SPDX-License-Identifier: Apache-2.0
package pagesarchive
import (
"fmt"
"path"
"path/filepath"
"strings"
)
// NormalizeEntryPath cleans an archive entry path and rejects zip-slip / absolute paths.
// skip=true means the entry should be ignored (empty path or directory marker).
func NormalizeEntryPath(raw string) (cleaned string, skip bool, err error) {
name := strings.TrimSpace(filepath.ToSlash(raw))
if name == "" {
return "", true, nil
}
if strings.HasSuffix(name, "/") {
return "", true, nil
}
if strings.HasPrefix(name, "/") || path.IsAbs(name) {
return "", false, fmt.Errorf("pages package contains absolute path: %s", raw)
}
// Reject Windows drive / UNC-style paths that may appear after ToSlash.
if len(name) >= 2 && name[1] == ':' {
return "", false, fmt.Errorf("pages package contains absolute path: %s", raw)
}
cleanedPath := path.Clean(name)
if cleanedPath == "." {
return "", true, nil
}
if cleanedPath == ".." || strings.HasPrefix(cleanedPath, "../") || strings.Contains(cleanedPath, "/../") {
return "", false, fmt.Errorf("pages package path escapes directory: %s", raw)
}
return cleanedPath, false, nil
}
// FindCommonRootPrefix returns a trailing-slash directory prefix shared by all file paths.
// When files do not share a single root folder the result is empty.
func FindCommonRootPrefix(paths []string) string {
var firstFilePath string
hasMultipleFiles := false
for _, item := range paths {
normalizedPath, skip, err := NormalizeEntryPath(item)
if err != nil || skip {
continue
}
if firstFilePath == "" {
firstFilePath = normalizedPath
} else {
hasMultipleFiles = true
}
}
if firstFilePath == "" {
return ""
}
parts := strings.Split(firstFilePath, "/")
if len(parts) <= 1 {
return ""
}
commonPrefix := parts[0] + "/"
if !hasMultipleFiles {
return commonPrefix
}
for _, item := range paths {
normalizedPath, skip, err := NormalizeEntryPath(item)
if err != nil || skip {
continue
}
if !strings.HasPrefix(normalizedPath, commonPrefix) {
return ""
}
}
return commonPrefix
}
// StripPrefix removes a directory prefix from a cleaned path when present.
func StripPrefix(normalizedPath, prefix string) string {
if prefix == "" {
return normalizedPath
}
return strings.TrimPrefix(normalizedPath, prefix)
}