# Drop-in for gitbay-runner.service, installed by `make deploy-runner` to # /etc/systemd/system/gitbay-runner.service.d/override.conf. # # A build must never starve the host: the e2e suite alone starts sixty # daemon instances, and with nothing holding it back a deploy's scp on # the admin sshd stalled at 1%. Lower CPU and IO weight keep sshd, # gitbayd and the backup timers responsive while a build runs. # # These weights are for the service, not per build, so `-jobs N` divides # them among N builds rather than taking N times as much. Raising -jobs # does not need them raised; it makes each build slower, not the host # busier. # # A build runs whatever the repository's ci.yml says, as the runner's # own user. Keep that user unprivileged: its key is added with # `keys add --scope runner`, which confines it to the runner protocol # and read-only git, and the sandboxing below keeps a step from # touching the system outside its workspace. # # Delegate=yes and the storage path below are what rootless podman needs # (#144): it manages its own cgroups for a container, and its image and # container store lives under the runner's home, which ProtectSystem # would otherwise make read-only. Prepare the host with # deploy/runner-podman-setup.sh before deploying a runner that isolates. [Service] # cmc/ci-smoke is the nightly isolation canary; a runner scoped to named # repositories never claims a build it is not scoped to, so the canary # must be listed or its scheduled build waits forever. # # Two layers of resource caps. MemoryMax and CPUQuota bound the unit — # the runner and every build together — which is what keeps the forge # alive when a build allocates without bound. -memory and -cpus on # ExecStart cap each build's own cgroup, which the runner creates under # this unit's delegated cgroup (Delegate=yes below); podman's own --memory # and --cpus never applied here, since under rootless cgroupfs the # container ran in the service cgroup itself (#188). # # 6G of the host's 7.7GB, no swap: the e2e suite peaks past 5GB, so the # cap sits above that rather than at a fair share. CPUQuota=300% and # -cpus 3 are three of the four cores, leaving one for gitbayd and sshd # (#144, #184). OOMPolicy=continue: systemd's default stops the whole # service when any process in it is OOM-killed, which would end the # runner mid-build; the build fails and the runner carries on. MemoryMax=6G CPUQuota=300% OOMPolicy=continue # # ExecStart is overridden here rather than left in the unit so the flags # and the sandboxing that has to match them live in one file: -isolation # podman needs NoNewPrivileges=no below, and -image needs an image the # host has been given (deploy/runner-podman-setup.sh, Containerfile.ci). # podman's run root is pinned under the runner's home by storage.conf # (runner-podman-setup.sh), not taken from XDG_RUNTIME_DIR or /tmp: this # unit has PrivateTmp, so a /tmp run root is a per-instance tmpfs. # # The cgroupfs manager puts podman's pause process under the user slice, # outside this unit's cgroup, so a stop does not end it and the next # start joins its namespaces — including a /tmp that no longer exists. # End it with the service. ExecStopPost=-/usr/bin/pkill -u ci-runner -x catatonit # On stop the runner drains: it claims nothing more and finishes the # build in flight, then exits. Give it long enough — the per-build limit # is 45m plus half a minute of report retries — before systemd kills it. # `make deploy-runner` therefore waits for a running build (#179). TimeoutStopSec=50min # The drain only works if the stop signal reaches the runner alone. The # default control-group mode sends SIGTERM to every process in the # cgroup at once — the ssh session streaming the log and the build's # container with it — so the runner drained a build whose steps were # already dead. mixed signals the main process only; whatever is left # when it exits is killed. KillMode=mixed # -untrusted: this runner isolates in podman, so it takes merge request # heads from forks; a runner without a container must not. ExecStart= ExecStart=/usr/local/bin/gitbay-runner -remote git@127.0.0.1 -workdir /var/lib/gitbay-runner/work -poll 5s -timeout 45m -repos krz/gitbay,cmc/ci-smoke -isolation podman -image localhost/gitbay-ci:1 -cpus 3 -memory 6g -untrusted Nice=10 CPUWeight=30 IOWeight=30 # NoNewPrivileges is off, and that is a deliberate trade (#144). # # Rootless podman sets up its user namespace with newuidmap, a setuid # helper; NoNewPrivileges=yes blocks it and podman fails with # "newuidmap: write to uid_map failed: Operation not permitted", so the # runner refuses to start. The choice is between this flag and running # builds in containers at all. # # Containers are the stronger boundary by a wide margin. NoNewPrivileges # constrained a process that was already executing arbitrary repository # code as this user; a container confines that code to an image and a # bind-mounted workspace. What is lost is one hardening layer on the # runner process itself, which is ours rather than a build's — a build no # longer runs in this process's context at all. # # Under -isolation none there is no container, and this flag should be # yes. Set it back if you run that way. NoNewPrivileges=no ProtectSystem=full # ProtectKernelTunables is off, for the same reason NoNewPrivileges is # (#144). It overmounts /proc/sys and friends in this unit's namespace, # and the kernel then refuses a fresh proc mount in any child user # namespace ("mount too revealing"): crun fails with "mount `proc` to # `proc`: Operation not permitted". There is no podman setting for it. # What the flag protected — /proc/sys from a build running on the host # as this user — the container now covers: a build gets its own proc, # with those paths masked by the runtime. Under -isolation none, set it # back to yes. # ProtectControlGroups is off because the runner writes cgroups: it # creates one per build under this unit's delegated cgroup to carry # -memory and -cpus (#188). The flag mounts /sys/fs/cgroup read-only in # the unit's namespace, which makes even a delegated cgroup unwritable, # and the runner then refuses to start when a limit is set. Delegate=yes # already hands this unit its subtree; what the flag protected beyond # that is other units' cgroups, which are root-owned and not writable # by this user regardless. ProtectControlGroups=no RestrictSUIDSGID=yes Delegate=yes # The runner's home is /var/lib/gitbay-runner (see the Admin page), and # the leading - makes a missing path ignored rather than fatal: this # drop-in installs on hosts that have not been prepared for podman yet, # and a unit that refuses to start would stop every build on the # instance. ReadWritePaths=-/var/lib/gitbay-runner/.local/share/containers -/var/lib/gitbay-runner/.config/containers