package ci import ( "archive/tar" "bytes" "context" "encoding/json" "errors" "fmt" "io" "maps" "net/http" "net/url" "os" "path" "slices" "strconv" "strings" "time" "hearthforge/internal/util" ) // Exit codes of vm/run-vm. const ( vmExitBuildFailed = 1 vmExitTooLarge = 3 ) // vmOverheadMB is the container memory on top of the VM's RAM, for QEMU // itself and the job tar. const vmOverheadMB = 512 func buildContainerName(runID int64) string { return fmt.Sprintf("hearthforge-ci-%d-build", runID) } // buildImage runs one build_image step in a build VM container, a sibling // of the CI container that boots QEMU from vm/. It returns the step log. // vars expands tags and args. It holds no secrets, because tags are public // and args end up in the image history. func (r *Runner) buildImage(ctx context.Context, runID int64, ciContainerID, repoName string, cfg *Config, spec *BuildImage, vars map[string]string, secrets []secret, onPartial func(string), ) (string, error) { if r.ImportImage == nil { return "", errors.New("no image store is configured") } tags, err := expandTags(spec.Tags, vars) if err != nil { return "", err } args := map[string]string{} for name, v := range spec.Args { args[name] = os.Expand(v, func(k string) string { return vars[k] }) } secretFiles := map[string]string{} for _, name := range spec.Secrets { i := slices.IndexFunc(secrets, func(s secret) bool { return s.name == name }) if i < 0 { return "", fmt.Errorf("secret %s is not set for this repository", name) } secretFiles[name] = secrets[i].value } contextPath := spec.Context if contextPath == "" { contextPath = cfg.CloneProjectTo } // The VM image build can take minutes, so its output goes to the step // log as it arrives. var setup logBuffer var lastFlush time.Time say := func(text string) { setup.append(text) if time.Since(lastFlush) >= 2*time.Second { onPartial(setup.String()) lastFlush = time.Now() } } vmImage, err := r.prepareVMImage(ctx, say) if err != nil { return setup.String(), err } id, err := r.createBuildContainer(ctx, runID, vmImage) if err != nil { return setup.String(), err } defer r.removeContainer(ctx, id) if err := r.uploadBuildJob(ctx, id, ciContainerID, contextPath, spec, args, secretFiles); err != nil { return setup.String(), err } if err := r.startContainer(ctx, id); err != nil { return setup.String(), err } logResp, err := r.do(ctx, http.MethodGet, "/containers/"+id+"/logs?follow=1&stdout=1&stderr=1", nil, "") if err != nil { return setup.String(), err } if logResp.StatusCode >= 300 { discard(logResp) return setup.String(), fmt.Errorf("cannot read the build VM log: HTTP %d", logResp.StatusCode) } buildLog := setup.String() + readMuxFrames(logResp.Body, func(partial string) { onPartial(setup.String() + partial) }) discard(logResp) if ctx.Err() != nil { return buildLog, ctx.Err() } code, err := r.waitContainer(ctx, id) if err != nil { return buildLog, err } switch code { case 0: case vmExitBuildFailed: return buildLog, errors.New("the build failed") case vmExitTooLarge: return buildLog, fmt.Errorf("the image is larger than CI_MAX_IMAGE_BYTES (%s)", formatBytes(r.cfg.CIMaxImageBytes)) default: return buildLog, fmt.Errorf("the build VM did not run (exit code %d)", code) } digest, err := r.importBuiltImage(ctx, id, repoName, spec.Image, tags) if err != nil { return buildLog, err } ref := vars["CI_REGISTRY"] if spec.Image != "" { ref += "/" + spec.Image } for _, tag := range tags { buildLog += fmt.Sprintf("Pushed %s:%s\n", ref, tag) } return buildLog + "Digest " + digest + "\n", nil } // expandTags expands each tag and drops the empty results, so a tag like // $CI_COMMIT_TAG can sit in a config that also runs on branches. func expandTags(tags []string, vars map[string]string) ([]string, error) { if len(tags) == 0 { tags = []string{"$CI_COMMIT_SHORT_SHA"} } var out []string for _, t := range tags { v := os.Expand(t, func(k string) string { return vars[k] }) if v == "" || slices.Contains(out, v) { continue } if !util.ValidImageTag(v) { return nil, fmt.Errorf("tag %q expands to %q, which is not a valid image tag", t, v) } out = append(out, v) } if len(out) == 0 { return nil, errors.New("every tag expanded to an empty string") } return out, nil } func (r *Runner) createBuildContainer(ctx context.Context, runID int64, image string) (string, error) { name := buildContainerName(runID) // A retry reuses the name, see createContainer. r.removeContainer(ctx, name) hostConfig := map[string]any{ "Devices": []map[string]string{ {"PathOnHost": "/dev/kvm", "PathInContainer": "/dev/kvm", "CgroupPermissions": "rwm"}, }, // Rootless Podman drops supplementary groups, and /dev/kvm is often // open to the kvm group only. crun reads this annotation, other // runtimes ignore it. "Annotations": map[string]string{"run.oci.keep_original_groups": "1"}, "Memory": (r.cfg.CIVMMemoryMB + vmOverheadMB) << 20, "NanoCpus": int64(r.cfg.CIVMCPUs) * 1e9, } if r.cfg.CINetwork != "" { hostConfig["NetworkMode"] = r.cfg.CINetwork } resp, body, err := r.doJSON(ctx, http.MethodPost, "/containers/create?name="+name, map[string]any{ "Image": image, "Env": []string{ "HF_VM_CPUS=" + strconv.Itoa(r.cfg.CIVMCPUs), "HF_VM_MEMORY_MB=" + strconv.FormatInt(r.cfg.CIVMMemoryMB, 10), "HF_VM_DISK_BYTES=" + strconv.FormatInt(r.cfg.CIVMDiskMB<<20, 10), "HF_VM_MAX_IMAGE_BYTES=" + strconv.FormatInt(r.cfg.CIMaxImageBytes, 10), }, "HostConfig": hostConfig, "Labels": ciContainerLabels, }) if err != nil { return "", err } if resp.StatusCode >= 300 { return "", fmt.Errorf("failed to create the build VM container: %d %s", resp.StatusCode, strings.TrimSpace(string(body))) } var data struct{ Id string } if err := json.Unmarshal(body, &data); err != nil { return "", err } return data.Id, nil } // uploadBuildJob fills /in of the build container: the context under // /in/context/, and the build settings as one file each // under /in/job. Files keep arbitrary values safe from shell quoting in the // guest. func (r *Runner) uploadBuildJob(ctx context.Context, buildID, ciContainerID, contextPath string, spec *BuildImage, args, secrets map[string]string, ) error { resp, err := r.do(ctx, http.MethodGet, "/containers/"+ciContainerID+"/archive?path="+url.QueryEscape(path.Clean(contextPath)), nil, "") if err != nil { return err } defer discard(resp) if resp.StatusCode >= 300 { return fmt.Errorf("cannot read the context %s: HTTP %d", contextPath, resp.StatusCode) } if err := r.putBuildArchive(ctx, buildID, "/in/context", resp.Body); err != nil { return err } files := map[string]string{} if spec.File != "" { files["job/file"] = spec.File } if spec.Target != "" { files["job/target"] = spec.Target } for name, v := range args { files["job/args/"+name] = v } for name, v := range secrets { files["job/secrets/"+name] = v } var job bytes.Buffer tw := tar.NewWriter(&job) for _, dir := range []string{"job/", "job/args/", "job/secrets/"} { if err := tw.WriteHeader(&tar.Header{Name: dir, Mode: 0o700, Typeflag: tar.TypeDir}); err != nil { return err } } for _, name := range slices.Sorted(maps.Keys(files)) { hdr := &tar.Header{Name: name, Mode: 0o600, Size: int64(len(files[name])), Typeflag: tar.TypeReg} if err := tw.WriteHeader(hdr); err != nil { return err } if _, err := io.WriteString(tw, files[name]); err != nil { return err } } if err := tw.Close(); err != nil { return err } return r.putBuildArchive(ctx, buildID, "/in", &job) } func (r *Runner) putBuildArchive(ctx context.Context, buildID, dest string, body io.Reader) error { resp, data, err := r.putArchive(ctx, buildID, dest, body) if err != nil { return err } if resp.StatusCode >= 300 { return fmt.Errorf("failed to upload the build job to %s: HTTP %d %s", dest, resp.StatusCode, strings.TrimSpace(string(data))) } return nil } // waitContainer waits for the container to stop and returns its exit code. func (r *Runner) waitContainer(ctx context.Context, id string) (int, error) { resp, body, err := r.doJSON(ctx, http.MethodPost, "/containers/"+id+"/wait", nil) if err != nil { return 0, err } if resp.StatusCode >= 300 { return 0, fmt.Errorf("wait for the build VM container: HTTP %d %s", resp.StatusCode, strings.TrimSpace(string(body))) } var data struct{ StatusCode int } if err := json.Unmarshal(body, &data); err != nil { return 0, err } return data.StatusCode, nil } // importBuiltImage streams /out/image.tar out of the stopped build // container into the registry. The archive endpoint wraps it in a tar of // its own. func (r *Runner) importBuiltImage(ctx context.Context, buildID, repoName, image string, tags []string) (string, error) { resp, err := r.do(ctx, http.MethodGet, "/containers/"+buildID+"/archive?path=/out/image.tar", nil, "") if err != nil { return "", err } defer discard(resp) if resp.StatusCode >= 300 { return "", fmt.Errorf("cannot read the built image: HTTP %d", resp.StatusCode) } tr := tar.NewReader(resp.Body) hdr, err := tr.Next() if err != nil { return "", fmt.Errorf("cannot read the built image: %w", err) } if hdr.Typeflag != tar.TypeReg { return "", errors.New("the built image is not a file") } // run-vm enforces the limit too. This check does not trust it. if hdr.Size > r.cfg.CIMaxImageBytes { return "", fmt.Errorf("the image is larger than CI_MAX_IMAGE_BYTES (%s)", formatBytes(r.cfg.CIMaxImageBytes)) } return r.ImportImage(ctx, repoName, image, tags, tr) }