cmd/gitbay-runner/cgroup_linux.go

v1.43.0
gitbay/cmd/gitbay-runner/cgroup_linux.go history · blame · raw

124 lines · 4315 bytes

  1//go:build linux
  2
  3package main
  4
  5import (
  6	"fmt"
  7	"log"
  8	"os"
  9	"os/exec"
 10	"path/filepath"
 11	"strconv"
 12	"syscall"
 13	"time"
 14)
 15
 16// buildCgroups is the runner's own cgroup subtree for builds. The unit's
 17// Delegate=yes hands the runner its service cgroup; the runner parks
 18// itself in a leaf so the service cgroup can enable controllers for
 19// children (a cgroup may hold processes or controller-enabled children,
 20// not both), and creates one child per build under builds/trusted or
 21// builds/untrusted.
 22type buildCgroups struct {
 23	builds string // <service cgroup>/builds
 24}
 25
 26const cgroupControllers = "+cpu +memory +pids"
 27
 28// prepareBuildCgroups moves the runner into <own>/runner, enables the
 29// controllers on its original cgroup, and creates builds/ with a child
 30// per trust class. The unit's drop-in may have created those before the
 31// runner started, to load the builds nftables table against them; they
 32// are used as found, never recreated, since the table holds their ids.
 33// It fails
 34// where the cgroup is not writable, which is a unit without
 35// Delegate=yes; the caller decides whether that is fatal.
 36func prepareBuildCgroups() (*buildCgroups, error) {
 37	raw, err := os.ReadFile("/proc/self/cgroup")
 38	if err != nil {
 39		return nil, err
 40	}
 41	own, err := ownCgroupPath(string(raw))
 42	if err != nil {
 43		return nil, err
 44	}
 45	root := filepath.Join("/sys/fs/cgroup", own)
 46	leaf := filepath.Join(root, "runner")
 47	if err := os.MkdirAll(leaf, 0o755); err != nil {
 48		return nil, fmt.Errorf("%s is not writable; the unit needs Delegate=yes: %w", root, err)
 49	}
 50	if err := os.WriteFile(filepath.Join(leaf, "cgroup.procs"), []byte(strconv.Itoa(os.Getpid())), 0o644); err != nil {
 51		return nil, fmt.Errorf("moving into %s: %w", leaf, err)
 52	}
 53	if err := os.WriteFile(filepath.Join(root, "cgroup.subtree_control"), []byte(cgroupControllers), 0o644); err != nil {
 54		return nil, fmt.Errorf("enabling controllers on %s: %w", root, err)
 55	}
 56	builds := filepath.Join(root, "builds")
 57	if err := os.MkdirAll(builds, 0o755); err != nil {
 58		return nil, err
 59	}
 60	if err := os.WriteFile(filepath.Join(builds, "cgroup.subtree_control"), []byte(cgroupControllers), 0o644); err != nil {
 61		return nil, fmt.Errorf("enabling controllers on %s: %w", builds, err)
 62	}
 63	for _, class := range buildClasses {
 64		dir := filepath.Join(builds, class)
 65		if err := os.MkdirAll(dir, 0o755); err != nil {
 66			return nil, err
 67		}
 68		if err := os.WriteFile(filepath.Join(dir, "cgroup.subtree_control"), []byte(cgroupControllers), 0o644); err != nil {
 69			return nil, fmt.Errorf("enabling controllers on %s: %w", dir, err)
 70		}
 71	}
 72	return &buildCgroups{builds: builds}, nil
 73}
 74
 75// create makes the cgroup for one build under its trust class with its
 76// limits written, and returns its path and an open directory fd for
 77// placing processes.
 78func (c *buildCgroups) create(id int64, trusted bool, memory, cpus string) (string, *os.File, error) {
 79	dir := buildCgroupDir(c.builds, id, trusted)
 80	if err := os.Mkdir(dir, 0o755); err != nil {
 81		return "", nil, err
 82	}
 83	if err := writeLimits(dir, memory, cpus); err != nil {
 84		os.Remove(dir)
 85		return "", nil, err
 86	}
 87	f, err := os.Open(dir)
 88	if err != nil {
 89		os.Remove(dir)
 90		return "", nil, err
 91	}
 92	return dir, f, nil
 93}
 94
 95// remove kills whatever is still in the build's cgroup — conmon, the
 96// pause process, a step's stray child — and removes it. rmdir fails
 97// until the kernel has reaped every process, so it retries briefly.
 98func (c *buildCgroups) remove(dir string) {
 99	if err := os.WriteFile(filepath.Join(dir, "cgroup.kill"), []byte("1"), 0o644); err != nil {
100		log.Printf("cgroup %s: kill: %v", dir, err)
101	}
102	for i := 0; i < 50; i++ {
103		if err := os.Remove(dir); err == nil {
104			return
105		}
106		time.Sleep(100 * time.Millisecond)
107	}
108	log.Printf("cgroup %s: still populated after kill; left in place", dir)
109}
110
111// intoCgroup starts cmd inside the cgroup fd refers to, so podman,
112// conmon and everything they start inherit the build's limits. A podman
113// exec started from the runner's own cgroup lands there, outside the
114// limit, which is why every invocation for a build goes through this.
115func intoCgroup(cmd *exec.Cmd, f *os.File) {
116	if f == nil {
117		return
118	}
119	if cmd.SysProcAttr == nil {
120		cmd.SysProcAttr = &syscall.SysProcAttr{}
121	}
122	cmd.SysProcAttr.UseCgroupFD = true
123	cmd.SysProcAttr.CgroupFD = int(f.Fd())
124}