cmd/gitbay-runner/cgroup_linux.go
124 lines · 4315 bytes
6 symbols in this file
1//go:build linux
2
3package main
4
5import (
6 "fmt"
7 "log"
8 "os"
9 "os/exec"
10 "path/filepath"
11 "strconv"
12 "syscall"
13 "time"
14)
15
16// buildCgroups is the runner's own cgroup subtree for builds. The unit's
17// Delegate=yes hands the runner its service cgroup; the runner parks
18// itself in a leaf so the service cgroup can enable controllers for
19// children (a cgroup may hold processes or controller-enabled children,
20// not both), and creates one child per build under builds/trusted or
21// builds/untrusted.
22type buildCgroups struct {
23 builds string // <service cgroup>/builds
24}
25
26const cgroupControllers = "+cpu +memory +pids"
27
28// prepareBuildCgroups moves the runner into <own>/runner, enables the
29// controllers on its original cgroup, and creates builds/ with a child
30// per trust class. The unit's drop-in may have created those before the
31// runner started, to load the builds nftables table against them; they
32// are used as found, never recreated, since the table holds their ids.
33// It fails
34// where the cgroup is not writable, which is a unit without
35// Delegate=yes; the caller decides whether that is fatal.
36func prepareBuildCgroups() (*buildCgroups, error) {
37 raw, err := os.ReadFile("/proc/self/cgroup")
38 if err != nil {
39 return nil, err
40 }
41 own, err := ownCgroupPath(string(raw))
42 if err != nil {
43 return nil, err
44 }
45 root := filepath.Join("/sys/fs/cgroup", own)
46 leaf := filepath.Join(root, "runner")
47 if err := os.MkdirAll(leaf, 0o755); err != nil {
48 return nil, fmt.Errorf("%s is not writable; the unit needs Delegate=yes: %w", root, err)
49 }
50 if err := os.WriteFile(filepath.Join(leaf, "cgroup.procs"), []byte(strconv.Itoa(os.Getpid())), 0o644); err != nil {
51 return nil, fmt.Errorf("moving into %s: %w", leaf, err)
52 }
53 if err := os.WriteFile(filepath.Join(root, "cgroup.subtree_control"), []byte(cgroupControllers), 0o644); err != nil {
54 return nil, fmt.Errorf("enabling controllers on %s: %w", root, err)
55 }
56 builds := filepath.Join(root, "builds")
57 if err := os.MkdirAll(builds, 0o755); err != nil {
58 return nil, err
59 }
60 if err := os.WriteFile(filepath.Join(builds, "cgroup.subtree_control"), []byte(cgroupControllers), 0o644); err != nil {
61 return nil, fmt.Errorf("enabling controllers on %s: %w", builds, err)
62 }
63 for _, class := range buildClasses {
64 dir := filepath.Join(builds, class)
65 if err := os.MkdirAll(dir, 0o755); err != nil {
66 return nil, err
67 }
68 if err := os.WriteFile(filepath.Join(dir, "cgroup.subtree_control"), []byte(cgroupControllers), 0o644); err != nil {
69 return nil, fmt.Errorf("enabling controllers on %s: %w", dir, err)
70 }
71 }
72 return &buildCgroups{builds: builds}, nil
73}
74
75// create makes the cgroup for one build under its trust class with its
76// limits written, and returns its path and an open directory fd for
77// placing processes.
78func (c *buildCgroups) create(id int64, trusted bool, memory, cpus string) (string, *os.File, error) {
79 dir := buildCgroupDir(c.builds, id, trusted)
80 if err := os.Mkdir(dir, 0o755); err != nil {
81 return "", nil, err
82 }
83 if err := writeLimits(dir, memory, cpus); err != nil {
84 os.Remove(dir)
85 return "", nil, err
86 }
87 f, err := os.Open(dir)
88 if err != nil {
89 os.Remove(dir)
90 return "", nil, err
91 }
92 return dir, f, nil
93}
94
95// remove kills whatever is still in the build's cgroup — conmon, the
96// pause process, a step's stray child — and removes it. rmdir fails
97// until the kernel has reaped every process, so it retries briefly.
98func (c *buildCgroups) remove(dir string) {
99 if err := os.WriteFile(filepath.Join(dir, "cgroup.kill"), []byte("1"), 0o644); err != nil {
100 log.Printf("cgroup %s: kill: %v", dir, err)
101 }
102 for i := 0; i < 50; i++ {
103 if err := os.Remove(dir); err == nil {
104 return
105 }
106 time.Sleep(100 * time.Millisecond)
107 }
108 log.Printf("cgroup %s: still populated after kill; left in place", dir)
109}
110
111// intoCgroup starts cmd inside the cgroup fd refers to, so podman,
112// conmon and everything they start inherit the build's limits. A podman
113// exec started from the runner's own cgroup lands there, outside the
114// limit, which is why every invocation for a build goes through this.
115func intoCgroup(cmd *exec.Cmd, f *os.File) {
116 if f == nil {
117 return
118 }
119 if cmd.SysProcAttr == nil {
120 cmd.SysProcAttr = &syscall.SysProcAttr{}
121 }
122 cmd.SysProcAttr.UseCgroupFD = true
123 cmd.SysProcAttr.CgroupFD = int(f.Fd())
124}