package ci import ( "archive/tar" "archive/zip" "bytes" "compress/gzip" "context" "errors" "fmt" "io" "net/http" "net/url" "os" "path" "strings" "github.com/klauspost/compress/zstd" ) // The container image needs no tar, gzip, zstd or zip. The engine streams // any path out of a container as a tar, and Hearthforge converts that // stream on the host. Directories are created the same way in reverse: a // tar with directory entries, extracted at /, needs no mkdir in the image. // mkdirTar builds a tar that creates dir and its parents when extracted at /. func mkdirTar(dir string) []byte { var buf bytes.Buffer tw := tar.NewWriter(&buf) clean := strings.TrimPrefix(path.Clean(dir), "/") if clean != "" && clean != "." { parts := strings.Split(clean, "/") for i := range parts { _ = tw.WriteHeader(&tar.Header{ Name: strings.Join(parts[:i+1], "/") + "/", Mode: 0o755, Typeflag: tar.TypeDir, }) } } _ = tw.Close() return buf.Bytes() } // mkdirInContainer creates dir and its parents without running a command. func (r *Runner) mkdirInContainer(ctx context.Context, containerID, dir string) error { resp, body, err := r.putArchive(ctx, containerID, "/", bytes.NewReader(mkdirTar(dir))) if err != nil { return err } if resp.StatusCode >= 300 { return fmt.Errorf("cannot create %s in the container: HTTP %d %s", dir, resp.StatusCode, strings.TrimSpace(string(body))) } return nil } // archiveFormats maps a publish_* option to its file extension. The order // is the order artifacts are collected in. var archiveFormats = []struct { pick func(Step) StringList ext string }{ {func(s Step) StringList { return s.PublishTar }, ".tar"}, {func(s Step) StringList { return s.PublishGzip }, ".tar.gz"}, {func(s Step) StringList { return s.PublishZstd }, ".tar.zst"}, {func(s Step) StringList { return s.PublishZip }, ".zip"}, } // storeArchive streams srcPath out of the container and writes it to // destPath in the format named by ext. maxBytes caps the written file. func (r *Runner) storeArchive(ctx context.Context, containerID, srcPath, destPath, ext string, maxBytes int64) (int64, error) { resp, err := r.do(ctx, http.MethodGet, "/containers/"+containerID+"/archive?path="+url.QueryEscape(srcPath), nil, "") if err != nil { return 0, err } defer discard(resp) if resp.StatusCode >= 300 { return 0, fmt.Errorf("cannot read %s: HTTP %d", srcPath, resp.StatusCode) } f, err := os.Create(destPath) if err != nil { return 0, err } lw := &limitedWriter{w: f, left: maxBytes} err = writeArchive(ext, resp.Body, lw) if cerr := f.Close(); err == nil { err = cerr } if err != nil { os.Remove(destPath) return 0, err } return maxBytes - lw.left, nil } // writeArchive converts a tar stream into the wanted format. func writeArchive(ext string, src io.Reader, dst io.Writer) error { switch ext { case ".tar": _, err := io.Copy(dst, src) return err case ".tar.gz": gz := gzip.NewWriter(dst) if _, err := io.Copy(gz, src); err != nil { return err } return gz.Close() case ".tar.zst": enc, err := zstd.NewWriter(dst) if err != nil { return err } if _, err := io.Copy(enc, src); err != nil { return err } return enc.Close() case ".zip": return tarToZip(src, dst) } return fmt.Errorf("unknown archive format %q", ext) } // tarToZip re-packs regular files and directories. Symlinks and devices // have no zip equivalent and are skipped. func tarToZip(src io.Reader, dst io.Writer) error { tr := tar.NewReader(src) zw := zip.NewWriter(dst) for { hdr, err := tr.Next() if errors.Is(err, io.EOF) { break } if err != nil { return err } switch hdr.Typeflag { case tar.TypeDir: if _, err := zw.CreateHeader(&zip.FileHeader{ Name: strings.TrimSuffix(hdr.Name, "/") + "/", Modified: hdr.ModTime, }); err != nil { return err } case tar.TypeReg: w, err := zw.CreateHeader(&zip.FileHeader{ Name: hdr.Name, Method: zip.Deflate, Modified: hdr.ModTime, }) if err != nil { return err } // The input tar is not compressed and the output is size-capped // by limitedWriter, so no decompression bomb is possible here. if _, err := io.Copy(w, tr); err != nil { //nolint:gosec return err } } } return zw.Close() } // errArtifactTooLarge is what a capped write reports. var errArtifactTooLarge = errors.New("artifact exceeds CI_MAX_ARTIFACT_BYTES") // limitedWriter fails once more than left bytes were written. type limitedWriter struct { w io.Writer left int64 } func (l *limitedWriter) Write(p []byte) (int, error) { if int64(len(p)) > l.left { return 0, errArtifactTooLarge } n, err := l.w.Write(p) l.left -= int64(n) return n, err }