cmd/gitbay-runner/main.go

672 lines · 23843 bytes

22 symbols in this file
  1// gitbay-runner executes CI builds queued by a gitbay server. It polls over
  2// SSH — the same authenticated channel everything else uses — claims one
  3// build at a time, clones the repo, runs each step with `sh -c`, streams the
  4// combined output back, and reports success or failure.
  5//
  6// The account behind the runner's key must be an instance admin: a runner
  7// executes arbitrary repo code, so handing out jobs is the operator's call.
  8// v1 runs steps directly on the host under this process's user; run it as a
  9// dedicated unprivileged user.
 10package main
 11
 12import (
 13	"encoding/json"
 14	"errors"
 15	"flag"
 16	"fmt"
 17	"io"
 18	"io/fs"
 19	"log"
 20	"net"
 21	"os"
 22	"os/exec"
 23	"os/signal"
 24	"path/filepath"
 25	"slices"
 26	"strings"
 27	"sync"
 28	"syscall"
 29	"time"
 30
 31	"gitbay.org/gitbay/internal/buildinfo"
 32	"gitbay.org/gitbay/internal/toolpath"
 33)
 34
 35type job struct {
 36	ID     int64    `json:"id"`
 37	Repo   string   `json:"repo"`
 38	Number int64    `json:"number"`
 39	Job    string   `json:"job"`
 40	SHA    string   `json:"sha"`
 41	Ref    string   `json:"ref"`
 42	Steps  []string `json:"steps"`
 43	Image  string   `json:"image"`
 44	// Trusted is false for a merge request head from a fork, and when the
 45	// server did not say: such a build gets no secrets and a home of its
 46	// own (#255).
 47	Trusted bool `json:"trusted"`
 48	// SSH is the instance's public ssh destination; a runner polling
 49	// over loopback takes its port for its builds (#260).
 50	SSH     string            `json:"ssh"`
 51	Secrets map[string]string `json:"secrets"`
 52}
 53
 54type runner struct {
 55	remote    string // ssh destination, e.g. git@gitbay.org
 56	sshOpts   []string
 57	cloneBase string // e.g. ssh://git@gitbay.org
 58	workdir   string
 59	timeout   time.Duration
 60	// image is the container image for a job that names none, and
 61	// isolation selects how steps run: "podman" or "none".
 62	image     string
 63	isolation string
 64	// memory and cpus cap one build's cgroup; empty means no cap.
 65	memory string
 66	cpus   string
 67	// cgroups is the runner's build cgroup subtree, nil where the unit
 68	// is not delegated and builds run unconfined in the service cgroup.
 69	cgroups *buildCgroups
 70	// stepFn is step, replaceable by tests.
 71	stepFn func() (bool, error)
 72	// repos limits which repositories this runner claims builds for. Empty
 73	// means any, which is what a runner on the server itself wants; a runner
 74	// somewhere that should not execute every repository's steps names them.
 75	repos []string
 76	// untrusted also claims merge request heads from forks.
 77	untrusted bool
 78}
 79
 80func main() {
 81	if len(os.Args) > 1 && os.Args[1] == "init" {
 82		os.Exit(runInit(os.Args[2:]))
 83	}
 84	var (
 85		configPath = flag.String("config", defaultConfigPath(), "config file; keys are these flag names, flags override it")
 86		identity   = flag.String("identity", "", "ssh private key to poll and clone with (default: the key gitbay-runner init generated, if present)")
 87		untrusted  = flag.Bool("untrusted", false, "also claim untrusted builds: merge request heads from forks (needs -isolation podman to be safe)")
 88		remote     = flag.String("remote", "git@gitbay.org", "ssh destination of the gitbay server")
 89		sshOpts    = flag.String("ssh-opts", "", "extra ssh options, space-separated (also used for git clone)")
 90		cloneBase  = flag.String("clone-base", "", "clone URL prefix (default ssh://<remote>)")
 91		workdir    = flag.String("workdir", defaultWorkdir(), "build workspace root")
 92		poll       = flag.Duration("poll", 5*time.Second, "idle poll interval")
 93		timeout    = flag.Duration("timeout", 30*time.Minute, "per-build time limit")
 94		repos      = flag.String("repos", "", "only claim builds for these repositories, comma-separated owner/name (default: any)")
 95		once       = flag.Bool("once", false, "process at most one build, then exit")
 96		jobs       = flag.Int("jobs", 1, "builds to run at once")
 97		image      = flag.String("image", "", "default container image for jobs that name none")
 98		isolation  = flag.String("isolation", "podman", "how steps run: podman, or none for no container")
 99		memory     = flag.String("memory", "", "memory limit per build, e.g. 4g (podman only, needs a delegated cgroup; default unlimited)")
100		cpus       = flag.String("cpus", "", "CPU limit per build, e.g. 2 (podman only, needs a delegated cgroup; default unlimited)")
101		version    = flag.Bool("version", false, "print the commit this binary was built from, then exit")
102	)
103	path := configPathFromArgs(os.Args[1:], *configPath)
104	if values, found, err := loadConfig(path); err != nil {
105		log.Fatal(err)
106	} else if found {
107		if err := applyConfig(flag.CommandLine, values); err != nil {
108			log.Fatal(err)
109		}
110		log.Printf("config: %s", path)
111	}
112	flag.Parse()
113	if *version {
114		fmt.Println(buildinfo.String())
115		return
116	}
117	// The runner links internal/store, so it goes stale on changes that never
118	// touch cmd/gitbay-runner. Say which commit is running.
119	log.Printf("gitbay-runner %s", buildinfo.String())
120	r := &runner{
121		remote:    *remote,
122		cloneBase: *cloneBase,
123		workdir:   *workdir,
124		timeout:   *timeout,
125		image:     *image,
126		isolation: *isolation,
127		memory:    *memory,
128		cpus:      *cpus,
129	}
130	var cgErr error
131	if r.isolation == isolationPodman {
132		// Before podman runs anything: its pause process lands in the
133		// cgroup of the first invocation, and that must be the runner's
134		// leaf, not a build's.
135		r.cgroups, cgErr = prepareBuildCgroups()
136	}
137	if err := r.checkIsolation(); err != nil {
138		// Refusing to start is the point. A runner that quietly fell back
139		// to running repository code on the host would drop isolation
140		// with nothing to surface it, which is worse than a stopped
141		// runner: the operator sees a failed unit either way, but only
142		// one of them is honest about why (#144).
143		log.Fatalf("isolation: %v", err)
144	}
145	if cgErr != nil {
146		// After checkIsolation, so a missing image or podman is reported
147		// as that rather than as the cgroups it would also lack.
148		if why := buildCgroupsRequired(r.memory, r.cpus, *untrusted, r.loopbackRemote()); why != "" {
149			log.Fatalf("%s: build cgroups unavailable: %v", why, cgErr)
150		}
151		log.Printf("build cgroups unavailable (%v); builds run unconfined in the service cgroup", cgErr)
152	}
153	if *sshOpts != "" {
154		r.sshOpts = strings.Fields(*sshOpts)
155	}
156	if *identity == "" {
157		if p := filepath.Join(configDir(), "id_ed25519"); fileExists(p) {
158			*identity = p
159		}
160	}
161	// Clipped: the later appends run from concurrent workers, and spare
162	// capacity here would have them writing the same backing array.
163	r.sshOpts = slices.Clip(sshOptions(*identity, r.sshOpts))
164	r.untrusted = *untrusted
165	for _, name := range strings.Split(*repos, ",") {
166		if name = strings.TrimSpace(name); name != "" {
167			r.repos = append(r.repos, name)
168		}
169	}
170	if r.cloneBase == "" {
171		r.cloneBase = "ssh://" + *remote
172	}
173	// 0o700, not 0o755: a build's checkout and its secrets-bearing
174	// environment are this user's business alone, and the default sits
175	// beside other users' data on a shared host.
176	if err := os.MkdirAll(r.workdir, 0o700); err != nil {
177		log.Fatal(err)
178	}
179	if err := checkWorkdir(r.workdir); err != nil {
180		log.Fatal(err)
181	}
182	n := *jobs
183	if n < 1 {
184		log.Fatal("-jobs must be at least 1")
185	}
186	if *once {
187		// "at most one build" is one build, whatever -jobs says.
188		n = 1
189	}
190	// `runner next` claims inside one transaction, so several workers
191	// claiming at once is already safe; the runner just never used that.
192	// Each build works in its own build-<id> directory, so they do not
193	// meet on disk either.
194	// A stop signal drains: no build is claimed after it, and each build
195	// already in flight runs to completion and is reported. The old
196	// behaviour was to die mid-build, which left the build "running" on
197	// the server with nothing executing (#179). The unit's
198	// TimeoutStopSec bounds the drain; a second signal ends it now.
199	stop := make(chan struct{})
200	go func() {
201		sigs := make(chan os.Signal, 2)
202		signal.Notify(sigs, syscall.SIGTERM, syscall.SIGINT)
203		<-sigs
204		log.Printf("draining: finishing builds in flight, claiming no more")
205		close(stop)
206		<-sigs
207		log.Printf("second signal: exiting without draining")
208		os.Exit(1)
209	}()
210	r.serve(n, *once, *poll, stop)
211}
212
213// serve runs n workers until stop closes. A worker checks stop only
214// between builds, so closing it never interrupts one.
215func (r *runner) serve(n int, once bool, poll time.Duration, stop <-chan struct{}) {
216	if r.stepFn == nil {
217		r.stepFn = r.step
218	}
219	var wg sync.WaitGroup
220	for i := 0; i < n; i++ {
221		wg.Add(1)
222		go func(i int) {
223			defer wg.Done()
224			if n > 1 {
225				time.Sleep(time.Duration(i) * poll / time.Duration(n))
226			}
227			for {
228				select {
229				case <-stop:
230					return
231				default:
232				}
233				ran, err := r.stepFn()
234				if err != nil {
235					log.Printf("runner: %v", err)
236				}
237				if once {
238					return
239				}
240				if !ran {
241					select {
242					case <-stop:
243						return
244					case <-time.After(poll):
245					}
246				}
247			}
248		}(i)
249	}
250	wg.Wait()
251}
252
253// step claims and executes at most one build. ran reports whether there was
254// one, so the caller knows when to idle.
255func (r *runner) step() (bool, error) {
256	args := []string{"runner", "next"}
257	if r.untrusted {
258		args = append(args, "--untrusted")
259	}
260	args = append(append(args, r.repos...), "--json")
261	out, err := r.ssh(nil, args...)
262	if err != nil {
263		return false, fmt.Errorf("claiming build: %w (%s)", err, out)
264	}
265	var env struct {
266		Data job `json:"data"`
267	}
268	if err := json.Unmarshal([]byte(out), &env); err != nil {
269		return false, fmt.Errorf("parsing job: %w", err)
270	}
271	if env.Data.ID == 0 {
272		return false, nil
273	}
274	j := env.Data
275	log.Printf("build %d: %s %s @ %.10s", j.ID, j.Repo, j.Job, j.SHA)
276	f := r.run(j)
277	status := "success"
278	if f != nil {
279		status = "failure"
280	}
281	if err := r.reportDone(j.ID, status, f); err != nil {
282		return true, err
283	}
284	log.Printf("build %d: %s", j.ID, status)
285	return true, nil
286}
287
288// logSink forwards a build's output to the server and swallows any error
289// doing so. os/exec surfaces a write failure on a step's stdout through
290// cmd.Wait(), so a sink that can fail is a sink that can fail the build it
291// was only recording — a restart or a dropped session used to turn a green
292// suite red, with the explaining line written to the same dead pipe. Losing
293// log lines is the acceptable failure here; losing the build is not.
294type logSink struct {
295	mu sync.Mutex
296	w  io.Writer // nil once a write has failed
297}
298
299func (s *logSink) Write(p []byte) (int, error) {
300	s.mu.Lock()
301	defer s.mu.Unlock()
302	if s.w != nil {
303		if _, err := s.w.Write(p); err != nil {
304			s.w = nil
305		}
306	}
307	return len(p), nil
308}
309
310// broken reports whether the stream was lost, so a build can say its log is
311// incomplete rather than appear to have simply stopped.
312func (s *logSink) broken() bool {
313	s.mu.Lock()
314	defer s.mu.Unlock()
315	return s.w == nil
316}
317
318// failure says where a build stopped: Step is the 1-based step that
319// failed, 0 when the build stopped before its first step (the clone, the
320// container), and Reason is one short line (#266).
321type failure struct {
322	Step   int
323	Reason string
324}
325
326// exitReason is how a finished command's failure reads in a build's log
327// and on the build: "exit 1" for a command that exited, the error
328// otherwise (a signal, a start failure).
329func exitReason(err error) string {
330	var ee *exec.ExitError
331	if errors.As(err, &ee) && ee.ExitCode() >= 0 {
332		return fmt.Sprintf("exit %d", ee.ExitCode())
333	}
334	return err.Error()
335}
336
337// run clones, checks out, and executes the steps, streaming output to the
338// server. Returns nil when every step succeeded, else where the build
339// stopped.
340func (r *runner) run(j job) *failure {
341	dir := filepath.Join(r.workdir, fmt.Sprintf("build-%d", j.ID))
342	defer os.RemoveAll(dir)
343
344	home, doneHome, err := buildHome(r.workdir, j)
345	if err != nil {
346		log.Printf("build %d: build home: %v", j.ID, err)
347		return &failure{Reason: "preparing the build home failed"}
348	}
349	defer doneHome()
350
351	// One long-lived `runner log` session receives the whole stream.
352	logCmd := exec.Command(toolpath.Look("ssh"), append(r.sshOpts, r.remote, "runner", "log", fmt.Sprint(j.ID))...)
353	pipe, err := logCmd.StdinPipe()
354	if err != nil {
355		log.Printf("build %d: log pipe: %v", j.ID, err)
356		return &failure{Reason: "opening the log stream failed"}
357	}
358	sink := &logSink{w: pipe}
359	logCmd.Stdout, logCmd.Stderr = io.Discard, io.Discard
360	if err := logCmd.Start(); err != nil {
361		log.Printf("build %d: log stream: %v", j.ID, err)
362		return &failure{Reason: "opening the log stream failed"}
363	}
364	// The server ends the log session with exit 3 when the build is
365	// cancelled; any other end is a lost stream, which the sink absorbs.
366	cancelled := make(chan struct{})
367	logExited := make(chan struct{})
368	go func() {
369		defer close(logExited)
370		err := logCmd.Wait()
371		if ee, ok := err.(*exec.ExitError); ok && ee.ExitCode() == 3 {
372			close(cancelled)
373			return
374		}
375		if err != nil {
376			log.Printf("build %d: log session ended: %v", j.ID, err)
377		}
378	}()
379	// runStep starts cmd and waits for it, the cancel signal, or the
380	// deadline. Every phase goes through it, so a cancel during the clone
381	// lands as fast as one during a step.
382	runStep := func(cmd *exec.Cmd, deadline time.Time) (bool, string) {
383		select {
384		case <-cancelled:
385			return false, "cancelled"
386		default:
387		}
388		ownProcessGroup(cmd)
389		if err := cmd.Start(); err != nil {
390			return false, fmt.Sprintf("start: %v", err)
391		}
392		done := make(chan error, 1)
393		go func() { done <- cmd.Wait() }()
394		// After a kill, Wait returns once every holder of the log pipe is
395		// gone; the group kill makes that prompt, and the cap makes sure a
396		// straggler cannot hold the build open.
397		reap := func() {
398			killTree(cmd)
399			select {
400			case <-done:
401			case <-time.After(10 * time.Second):
402			}
403		}
404		select {
405		case err := <-done:
406			if err != nil {
407				return false, exitReason(err)
408			}
409			return true, ""
410		case <-cancelled:
411			reap()
412			return false, "cancelled"
413		case <-time.After(time.Until(deadline)):
414			reap()
415			return false, fmt.Sprintf("build timed out after %s", r.timeout)
416		}
417	}
418	defer func() {
419		select {
420		case <-cancelled:
421			log.Printf("build %d: cancelled", j.ID)
422		default:
423			if sink.broken() {
424				log.Printf("build %d: log stream lost; stored log is incomplete", j.ID)
425			}
426		}
427		pipe.Close()
428		<-logExited
429	}()
430
431	gitSSH := strings.TrimSpace("ssh " + strings.Join(r.sshOpts, " "))
432	cloneURL := r.cloneBase + "/" + j.Repo + ".git"
433	deadline := time.Now().Add(r.timeout)
434	fmt.Fprintf(sink, "$ git clone %s (%.10s)\n", cloneURL, j.SHA)
435	// A merge request head lives under refs/merge-requests/, which a
436	// clone does not fetch; ask for the ref before checking out.
437	steps := [][]string{{"clone", "-q", cloneURL, dir}}
438	if strings.HasPrefix(j.Ref, "refs/") {
439		steps = append(steps, []string{"-C", dir, "fetch", "-q", "origin", j.Ref})
440	}
441	steps = append(steps, []string{"-C", dir, "checkout", "-q", j.SHA})
442	for _, args := range steps {
443		cmd := exec.Command(toolpath.Look("git"), args...)
444		cmd.Env = append(os.Environ(), "GIT_SSH_COMMAND="+gitSSH, "GIT_TERMINAL_PROMPT=0")
445		cmd.Stdout, cmd.Stderr = sink, sink
446		if ok, why := runStep(cmd, deadline); !ok {
447			fmt.Fprintf(sink, "git %s: %s\n", args[0], why)
448			return &failure{Reason: "git " + args[0] + ": " + why}
449		}
450	}
451
452	env := stepEnv(j, home, r.buildSSH(j.SSH))
453	return r.runSteps(j, dir, env, sink, deadline, runStep)
454}
455
456// buildHome is a build's HOME and what to do with it when the build ends.
457//
458// Not the workspace, which is removed after every build: the Go module
459// cache and every other tool cache live under HOME. Not the runner's own
460// home either, where its SSH key and credential dotfiles are.
461//
462// A trusted build gets its repository's home,
463// <workdir>/trusted-home/<owner>/<name>, kept between builds so the
464// caches survive. One per repository: shared across repositories, a step
465// could poison a cache or plant a .gitconfig that another repository's
466// build would honour (#184). The root is not <workdir>/home, where homes
467// that untrusted builds could write were kept before #255, so none of
468// those is read again.
469//
470// An untrusted build gets <workdir>/build-<id>-home, new and empty,
471// removed when the build ends. The container mounts HOME read-write, so
472// a home a fork's build could write is a cache a stranger controls
473// (#255).
474func buildHome(workdir string, j job) (string, func(), error) {
475	if !j.Trusted {
476		dir := filepath.Join(workdir, fmt.Sprintf("build-%d-home", j.ID))
477		if err := os.Mkdir(dir, 0o700); err != nil {
478			return "", nil, err
479		}
480		return dir, func() {
481			if err := removeTree(dir); err != nil {
482				log.Printf("build %d: removing its home: %v", j.ID, err)
483			}
484		}, nil
485	}
486	root := filepath.Join(workdir, "trusted-home")
487	dir := filepath.Join(root, filepath.FromSlash(j.Repo))
488	if rel, err := filepath.Rel(root, dir); err != nil || rel == "." || strings.HasPrefix(rel, "..") {
489		return "", nil, fmt.Errorf("repository path %q escapes the build home root", j.Repo)
490	}
491	if err := os.MkdirAll(dir, 0o700); err != nil {
492		return "", nil, err
493	}
494	return dir, func() {}, nil
495}
496
497// removeTree deletes dir and everything under it. os.RemoveAll alone
498// fails on a directory without write permission, and the Go module cache
499// makes every directory it fills read-only.
500func removeTree(dir string) error {
501	filepath.WalkDir(dir, func(p string, d fs.DirEntry, err error) error {
502		if err == nil && d.IsDir() {
503			os.Chmod(p, 0o700)
504		}
505		return nil
506	})
507	return os.RemoveAll(dir)
508}
509
510// loopbackRemote reports whether the runner polls the daemon on its own
511// host over loopback.
512func (r *runner) loopbackRemote() bool {
513	_, host, ok := strings.Cut(r.remote, "@")
514	if !ok {
515		host = r.remote
516	}
517	return host == "127.0.0.1" || host == "localhost" || host == "::1"
518}
519
520// hostAddr is the address a podman build under pasta reaches the host
521// at: pasta's --map-guest-addr, which podman sets to this address and
522// which pasta translates to the host's public address.
523const hostAddr = "169.254.1.2"
524
525// buildSSH is the instance's ssh destination as a build reaches it.
526// Under pasta a podman build holds the host's own public address, so the
527// instance's public name resolves to the container itself. A runner
528// polling over loopback runs on the daemon's host, and its podman builds
529// get hostAddr with the user from -remote and the port from the claim's
530// destination when it is not 22. Any other runner's -remote is the path
531// that reaches the forge from where it runs, so its builds get that.
532func (r *runner) buildSSH(public string) string {
533	if r.isolation == isolationPodman && r.loopbackRemote() {
534		user, _, ok := strings.Cut(r.remote, "@")
535		if !ok || user == "" {
536			user = "git"
537		}
538		dest := hostAddr
539		_, hostport, ok := strings.Cut(public, "@")
540		if !ok {
541			hostport = public
542		}
543		if _, port, err := net.SplitHostPort(hostport); err == nil && port != "" && port != "22" {
544			dest = net.JoinHostPort(hostAddr, port)
545		}
546		return user + "@" + dest
547	}
548	return r.remote
549}
550
551// buildNetwork is the podman network option for a build on the daemon's
552// host. A build reaches the host at hostAddr, which pasta translates to
553// the host's public address, and cannot reach the host's loopback, so
554// none of its connections arrive from 127.0.0.1, the address the runner
555// polls from; the SSH auth limiter counts failures per source address
556// (#260). podman passes --no-map-gw to pasta by default; it is stated
557// here so the build's view of the host does not depend on that default.
558// The host's nftables tables (deploy/gitbay-runner-egress.nft and
559// gitbay-runner-builds.nft) limit what a build reaches on the host to
560// 22, 80 and 443, and an untrusted build to nothing but DNS.
561func (r *runner) buildNetwork() []string {
562	if !r.loopbackRemote() {
563		return nil
564	}
565	return []string{"--network", "pasta:--no-map-gw"}
566}
567
568// stepEnv builds the environment a build step runs with. It is
569// constructed, not inherited: os.Environ() would hand repository content
570// the runner's entire environment, including anything an operator set on
571// the service (#144).
572//
573// HOME is the build's home (buildHome), not the runner's own: tools read
574// credentials out of dotfiles — .netrc, .npmrc, .gitconfig — and a build
575// has no business finding the runner's.
576//
577// PATH is the one thing carried over: without it a step cannot find the
578// tools the host was provisioned with.
579func stepEnv(j job, home, sshDest string) []string {
580	path := os.Getenv("PATH")
581	if path == "" {
582		path = "/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin"
583	}
584	env := []string{
585		"PATH=" + path,
586		"HOME=" + home,
587		"LANG=C.UTF-8",
588		"CI=true",
589		"GITBAY_REPO=" + j.Repo,
590		"GITBAY_SHA=" + j.SHA,
591		"GITBAY_REF=" + j.Ref,
592		"GITBAY_JOB=" + j.Job,
593		"GITBAY_SSH=" + sshDest,
594	}
595	// The server sends secrets only for a trusted build. The claim's
596	// trust flag decides here as well, not whether any arrived (#255).
597	if j.Trusted {
598		for name, value := range j.Secrets {
599			env = append(env, name+"="+value)
600		}
601	}
602	return env
603}
604
605// ssh runs one control command against the server and returns stdout.// ssh runs one control command against the server and returns stdout.
606func (r *runner) ssh(stdin io.Reader, args ...string) (string, error) {
607	cmd := exec.Command(toolpath.Look("ssh"), append(append(r.sshOpts, r.remote), args...)...)
608	if stdin != nil {
609		cmd.Stdin = stdin
610	}
611	var out, errOut strings.Builder
612	cmd.Stdout, cmd.Stderr = &out, &errOut
613	if err := cmd.Run(); err != nil {
614		return out.String() + errOut.String(), err
615	}
616	return out.String(), nil
617}
618
619func fileExists(p string) bool { _, err := os.Stat(p); return err == nil }
620
621// defaultWorkdir picks a build workspace that another local user cannot
622// have created first.
623//
624// The default used to be <tmp>/gitbay-runner: a fixed name inside a
625// world-writable directory, created with MkdirAll, which succeeds against
626// an existing directory whoever owns it. On a shared host another user
627// could have made it — or symlinked it — before the runner started, and
628// this is the process that clones repositories and exports build secrets
629// into step environments (go:S5445, #153).
630//
631// The user's cache directory is not world-writable and is per-user by
632// construction. Falling back to tmp keeps a runner working where HOME is
633// unset, and checkWorkdir refuses the unsafe cases there.
634func defaultWorkdir() string {
635	if cache, err := os.UserCacheDir(); err == nil && cache != "" {
636		return filepath.Join(cache, "gitbay-runner")
637	}
638	return filepath.Join(os.TempDir(), "gitbay-runner")
639}
640
641// checkWorkdir makes sure the workspace is a directory this user owns
642// privately. MkdirAll is happy with one that already exists, so being
643// able to create it proves nothing about who made it.
644//
645// A directory we own that is merely too permissive is tightened rather
646// than refused: every runner before this one created its workspace 0755,
647// so refusing would take the runner down on upgrade to fix a permission
648// we are entitled to change. What cannot be repaired — a symlink, or
649// something owned by someone else — is refused, because those are what an
650// attacker leaves behind and neither is ours to correct.
651func checkWorkdir(dir string) error {
652	fi, err := os.Lstat(dir)
653	if err != nil {
654		return err
655	}
656	if fi.Mode()&os.ModeSymlink != 0 {
657		return fmt.Errorf("workdir %s is a symlink; point -workdir at a real directory", dir)
658	}
659	if !fi.IsDir() {
660		return fmt.Errorf("workdir %s is not a directory", dir)
661	}
662	if st, ok := fi.Sys().(*syscall.Stat_t); ok && int(st.Uid) != os.Getuid() {
663		return fmt.Errorf("workdir %s is owned by uid %d, not this process's %d", dir, st.Uid, os.Getuid())
664	}
665	if perm := fi.Mode().Perm(); perm&0o077 != 0 {
666		log.Printf("workdir %s was mode %04o; tightening to 0700 (builds and their secrets are this user's alone)", dir, perm)
667		if err := os.Chmod(dir, 0o700); err != nil {
668			return fmt.Errorf("tightening workdir %s: %w", dir, err)
669		}
670	}
671	return nil
672}