cmd/gitbay-runner/main.go
673 lines · 23738 bytes
1// gitbay-runner executes CI builds queued by a gitbay server. It polls over
2// SSH — the same authenticated channel everything else uses — claims one
3// build at a time, clones the repo, runs each step with `sh -c`, streams the
4// combined output back, and reports success or failure.
5//
6// The account behind the runner's key must be an instance admin: a runner
7// executes arbitrary repo code, so handing out jobs is the operator's call.
8// v1 runs steps directly on the host under this process's user; run it as a
9// dedicated unprivileged user.
10package main
11
12import (
13 "encoding/json"
14 "errors"
15 "flag"
16 "fmt"
17 "io"
18 "io/fs"
19 "log"
20 "net"
21 "os"
22 "os/exec"
23 "os/signal"
24 "path/filepath"
25 "slices"
26 "strings"
27 "sync"
28 "syscall"
29 "time"
30
31 "gitbay.org/gitbay/internal/buildinfo"
32 "gitbay.org/gitbay/internal/toolpath"
33)
34
35type job struct {
36 ID int64 `json:"id"`
37 Repo string `json:"repo"`
38 Number int64 `json:"number"`
39 Job string `json:"job"`
40 SHA string `json:"sha"`
41 Ref string `json:"ref"`
42 Steps []string `json:"steps"`
43 Image string `json:"image"`
44 // Trusted is false for a merge request head from a fork, and when the
45 // server did not say: such a build gets no secrets and a home of its
46 // own (#255).
47 Trusted bool `json:"trusted"`
48 // SSH is the instance's public ssh destination; a runner polling
49 // over loopback takes its port for its builds (#260).
50 SSH string `json:"ssh"`
51 Secrets map[string]string `json:"secrets"`
52}
53
54type runner struct {
55 remote string // ssh destination, e.g. git@gitbay.org
56 sshOpts []string
57 cloneBase string // e.g. ssh://git@gitbay.org
58 workdir string
59 timeout time.Duration
60 // image is the container image for a job that names none, and
61 // isolation selects how steps run: "podman" or "none".
62 image string
63 isolation string
64 // memory and cpus cap one build's cgroup; empty means no cap.
65 memory string
66 cpus string
67 // cgroups is the runner's build cgroup subtree, nil where the unit
68 // is not delegated and builds run unconfined in the service cgroup.
69 cgroups *buildCgroups
70 // stepFn is step, replaceable by tests.
71 stepFn func() (bool, error)
72 // repos limits which repositories this runner claims builds for. Empty
73 // means any, which is what a runner on the server itself wants; a runner
74 // somewhere that should not execute every repository's steps names them.
75 repos []string
76 // untrusted also claims merge request heads from forks.
77 untrusted bool
78}
79
80func main() {
81 if len(os.Args) > 1 && os.Args[1] == "init" {
82 os.Exit(runInit(os.Args[2:]))
83 }
84 var (
85 configPath = flag.String("config", defaultConfigPath(), "config file; keys are these flag names, flags override it")
86 identity = flag.String("identity", "", "ssh private key to poll and clone with (default: the key gitbay-runner init generated, if present)")
87 untrusted = flag.Bool("untrusted", false, "also claim untrusted builds: merge request heads from forks (needs -isolation podman to be safe)")
88 remote = flag.String("remote", "git@gitbay.org", "ssh destination of the gitbay server")
89 sshOpts = flag.String("ssh-opts", "", "extra ssh options, space-separated (also used for git clone)")
90 cloneBase = flag.String("clone-base", "", "clone URL prefix (default ssh://<remote>)")
91 workdir = flag.String("workdir", defaultWorkdir(), "build workspace root")
92 poll = flag.Duration("poll", 5*time.Second, "idle poll interval")
93 timeout = flag.Duration("timeout", 30*time.Minute, "per-build time limit")
94 repos = flag.String("repos", "", "only claim builds for these repositories, comma-separated owner/name (default: any)")
95 once = flag.Bool("once", false, "process at most one build, then exit")
96 jobs = flag.Int("jobs", 1, "builds to run at once")
97 image = flag.String("image", "", "default container image for jobs that name none")
98 isolation = flag.String("isolation", "podman", "how steps run: podman, or none for no container")
99 memory = flag.String("memory", "", "memory limit per build, e.g. 4g (podman only, needs a delegated cgroup; default unlimited)")
100 cpus = flag.String("cpus", "", "CPU limit per build, e.g. 2 (podman only, needs a delegated cgroup; default unlimited)")
101 version = flag.Bool("version", false, "print the commit this binary was built from, then exit")
102 )
103 path := configPathFromArgs(os.Args[1:], *configPath)
104 if values, found, err := loadConfig(path); err != nil {
105 log.Fatal(err)
106 } else if found {
107 if err := applyConfig(flag.CommandLine, values); err != nil {
108 log.Fatal(err)
109 }
110 log.Printf("config: %s", path)
111 }
112 flag.Parse()
113 if *version {
114 fmt.Println(buildinfo.String())
115 return
116 }
117 // The runner links internal/store, so it goes stale on changes that never
118 // touch cmd/gitbay-runner. Say which commit is running.
119 log.Printf("gitbay-runner %s", buildinfo.String())
120 r := &runner{
121 remote: *remote,
122 cloneBase: *cloneBase,
123 workdir: *workdir,
124 timeout: *timeout,
125 image: *image,
126 isolation: *isolation,
127 memory: *memory,
128 cpus: *cpus,
129 }
130 if r.isolation == isolationPodman {
131 // Before podman runs anything: its pause process lands in the
132 // cgroup of the first invocation, and that must be the runner's
133 // leaf, not a build's.
134 cg, err := prepareBuildCgroups()
135 switch {
136 case err == nil:
137 r.cgroups = cg
138 case r.memory != "" || r.cpus != "":
139 // Limits that cannot be applied are refused, not dropped:
140 // a runner that accepted -memory and ran uncapped is what
141 // #188 was.
142 log.Fatalf("-memory/-cpus: build cgroups unavailable: %v", err)
143 default:
144 log.Printf("build cgroups unavailable (%v); builds run unconfined in the service cgroup", err)
145 }
146 }
147 if err := r.checkIsolation(); err != nil {
148 // Refusing to start is the point. A runner that quietly fell back
149 // to running repository code on the host would drop isolation
150 // with nothing to surface it, which is worse than a stopped
151 // runner: the operator sees a failed unit either way, but only
152 // one of them is honest about why (#144).
153 log.Fatalf("isolation: %v", err)
154 }
155 if *sshOpts != "" {
156 r.sshOpts = strings.Fields(*sshOpts)
157 }
158 if *identity == "" {
159 if p := filepath.Join(configDir(), "id_ed25519"); fileExists(p) {
160 *identity = p
161 }
162 }
163 // Clipped: the later appends run from concurrent workers, and spare
164 // capacity here would have them writing the same backing array.
165 r.sshOpts = slices.Clip(sshOptions(*identity, r.sshOpts))
166 r.untrusted = *untrusted
167 for _, name := range strings.Split(*repos, ",") {
168 if name = strings.TrimSpace(name); name != "" {
169 r.repos = append(r.repos, name)
170 }
171 }
172 if r.cloneBase == "" {
173 r.cloneBase = "ssh://" + *remote
174 }
175 // 0o700, not 0o755: a build's checkout and its secrets-bearing
176 // environment are this user's business alone, and the default sits
177 // beside other users' data on a shared host.
178 if err := os.MkdirAll(r.workdir, 0o700); err != nil {
179 log.Fatal(err)
180 }
181 if err := checkWorkdir(r.workdir); err != nil {
182 log.Fatal(err)
183 }
184 n := *jobs
185 if n < 1 {
186 log.Fatal("-jobs must be at least 1")
187 }
188 if *once {
189 // "at most one build" is one build, whatever -jobs says.
190 n = 1
191 }
192 // `runner next` claims inside one transaction, so several workers
193 // claiming at once is already safe; the runner just never used that.
194 // Each build works in its own build-<id> directory, so they do not
195 // meet on disk either.
196 // A stop signal drains: no build is claimed after it, and each build
197 // already in flight runs to completion and is reported. The old
198 // behaviour was to die mid-build, which left the build "running" on
199 // the server with nothing executing (#179). The unit's
200 // TimeoutStopSec bounds the drain; a second signal ends it now.
201 stop := make(chan struct{})
202 go func() {
203 sigs := make(chan os.Signal, 2)
204 signal.Notify(sigs, syscall.SIGTERM, syscall.SIGINT)
205 <-sigs
206 log.Printf("draining: finishing builds in flight, claiming no more")
207 close(stop)
208 <-sigs
209 log.Printf("second signal: exiting without draining")
210 os.Exit(1)
211 }()
212 r.serve(n, *once, *poll, stop)
213}
214
215// serve runs n workers until stop closes. A worker checks stop only
216// between builds, so closing it never interrupts one.
217func (r *runner) serve(n int, once bool, poll time.Duration, stop <-chan struct{}) {
218 if r.stepFn == nil {
219 r.stepFn = r.step
220 }
221 var wg sync.WaitGroup
222 for i := 0; i < n; i++ {
223 wg.Add(1)
224 go func(i int) {
225 defer wg.Done()
226 if n > 1 {
227 time.Sleep(time.Duration(i) * poll / time.Duration(n))
228 }
229 for {
230 select {
231 case <-stop:
232 return
233 default:
234 }
235 ran, err := r.stepFn()
236 if err != nil {
237 log.Printf("runner: %v", err)
238 }
239 if once {
240 return
241 }
242 if !ran {
243 select {
244 case <-stop:
245 return
246 case <-time.After(poll):
247 }
248 }
249 }
250 }(i)
251 }
252 wg.Wait()
253}
254
255// step claims and executes at most one build. ran reports whether there was
256// one, so the caller knows when to idle.
257func (r *runner) step() (bool, error) {
258 args := []string{"runner", "next"}
259 if r.untrusted {
260 args = append(args, "--untrusted")
261 }
262 args = append(append(args, r.repos...), "--json")
263 out, err := r.ssh(nil, args...)
264 if err != nil {
265 return false, fmt.Errorf("claiming build: %w (%s)", err, out)
266 }
267 var env struct {
268 Data job `json:"data"`
269 }
270 if err := json.Unmarshal([]byte(out), &env); err != nil {
271 return false, fmt.Errorf("parsing job: %w", err)
272 }
273 if env.Data.ID == 0 {
274 return false, nil
275 }
276 j := env.Data
277 log.Printf("build %d: %s %s @ %.10s", j.ID, j.Repo, j.Job, j.SHA)
278 f := r.run(j)
279 status := "success"
280 if f != nil {
281 status = "failure"
282 }
283 if err := r.reportDone(j.ID, status, f); err != nil {
284 return true, err
285 }
286 log.Printf("build %d: %s", j.ID, status)
287 return true, nil
288}
289
290// logSink forwards a build's output to the server and swallows any error
291// doing so. os/exec surfaces a write failure on a step's stdout through
292// cmd.Wait(), so a sink that can fail is a sink that can fail the build it
293// was only recording — a restart or a dropped session used to turn a green
294// suite red, with the explaining line written to the same dead pipe. Losing
295// log lines is the acceptable failure here; losing the build is not.
296type logSink struct {
297 mu sync.Mutex
298 w io.Writer // nil once a write has failed
299}
300
301func (s *logSink) Write(p []byte) (int, error) {
302 s.mu.Lock()
303 defer s.mu.Unlock()
304 if s.w != nil {
305 if _, err := s.w.Write(p); err != nil {
306 s.w = nil
307 }
308 }
309 return len(p), nil
310}
311
312// broken reports whether the stream was lost, so a build can say its log is
313// incomplete rather than appear to have simply stopped.
314func (s *logSink) broken() bool {
315 s.mu.Lock()
316 defer s.mu.Unlock()
317 return s.w == nil
318}
319
320// failure says where a build stopped: Step is the 1-based step that
321// failed, 0 when the build stopped before its first step (the clone, the
322// container), and Reason is one short line (#266).
323type failure struct {
324 Step int
325 Reason string
326}
327
328// exitReason is how a finished command's failure reads in a build's log
329// and on the build: "exit 1" for a command that exited, the error
330// otherwise (a signal, a start failure).
331func exitReason(err error) string {
332 var ee *exec.ExitError
333 if errors.As(err, &ee) && ee.ExitCode() >= 0 {
334 return fmt.Sprintf("exit %d", ee.ExitCode())
335 }
336 return err.Error()
337}
338
339// run clones, checks out, and executes the steps, streaming output to the
340// server. Returns nil when every step succeeded, else where the build
341// stopped.
342func (r *runner) run(j job) *failure {
343 dir := filepath.Join(r.workdir, fmt.Sprintf("build-%d", j.ID))
344 defer os.RemoveAll(dir)
345
346 home, doneHome, err := buildHome(r.workdir, j)
347 if err != nil {
348 log.Printf("build %d: build home: %v", j.ID, err)
349 return &failure{Reason: "preparing the build home failed"}
350 }
351 defer doneHome()
352
353 // One long-lived `runner log` session receives the whole stream.
354 logCmd := exec.Command(toolpath.Look("ssh"), append(r.sshOpts, r.remote, "runner", "log", fmt.Sprint(j.ID))...)
355 pipe, err := logCmd.StdinPipe()
356 if err != nil {
357 log.Printf("build %d: log pipe: %v", j.ID, err)
358 return &failure{Reason: "opening the log stream failed"}
359 }
360 sink := &logSink{w: pipe}
361 logCmd.Stdout, logCmd.Stderr = io.Discard, io.Discard
362 if err := logCmd.Start(); err != nil {
363 log.Printf("build %d: log stream: %v", j.ID, err)
364 return &failure{Reason: "opening the log stream failed"}
365 }
366 // The server ends the log session with exit 3 when the build is
367 // cancelled; any other end is a lost stream, which the sink absorbs.
368 cancelled := make(chan struct{})
369 logExited := make(chan struct{})
370 go func() {
371 defer close(logExited)
372 err := logCmd.Wait()
373 if ee, ok := err.(*exec.ExitError); ok && ee.ExitCode() == 3 {
374 close(cancelled)
375 return
376 }
377 if err != nil {
378 log.Printf("build %d: log session ended: %v", j.ID, err)
379 }
380 }()
381 // runStep starts cmd and waits for it, the cancel signal, or the
382 // deadline. Every phase goes through it, so a cancel during the clone
383 // lands as fast as one during a step.
384 runStep := func(cmd *exec.Cmd, deadline time.Time) (bool, string) {
385 select {
386 case <-cancelled:
387 return false, "cancelled"
388 default:
389 }
390 ownProcessGroup(cmd)
391 if err := cmd.Start(); err != nil {
392 return false, fmt.Sprintf("start: %v", err)
393 }
394 done := make(chan error, 1)
395 go func() { done <- cmd.Wait() }()
396 // After a kill, Wait returns once every holder of the log pipe is
397 // gone; the group kill makes that prompt, and the cap makes sure a
398 // straggler cannot hold the build open.
399 reap := func() {
400 killTree(cmd)
401 select {
402 case <-done:
403 case <-time.After(10 * time.Second):
404 }
405 }
406 select {
407 case err := <-done:
408 if err != nil {
409 return false, exitReason(err)
410 }
411 return true, ""
412 case <-cancelled:
413 reap()
414 return false, "cancelled"
415 case <-time.After(time.Until(deadline)):
416 reap()
417 return false, fmt.Sprintf("build timed out after %s", r.timeout)
418 }
419 }
420 defer func() {
421 select {
422 case <-cancelled:
423 log.Printf("build %d: cancelled", j.ID)
424 default:
425 if sink.broken() {
426 log.Printf("build %d: log stream lost; stored log is incomplete", j.ID)
427 }
428 }
429 pipe.Close()
430 <-logExited
431 }()
432
433 gitSSH := strings.TrimSpace("ssh " + strings.Join(r.sshOpts, " "))
434 cloneURL := r.cloneBase + "/" + j.Repo + ".git"
435 deadline := time.Now().Add(r.timeout)
436 fmt.Fprintf(sink, "$ git clone %s (%.10s)\n", cloneURL, j.SHA)
437 // A merge request head lives under refs/merge-requests/, which a
438 // clone does not fetch; ask for the ref before checking out.
439 steps := [][]string{{"clone", "-q", cloneURL, dir}}
440 if strings.HasPrefix(j.Ref, "refs/") {
441 steps = append(steps, []string{"-C", dir, "fetch", "-q", "origin", j.Ref})
442 }
443 steps = append(steps, []string{"-C", dir, "checkout", "-q", j.SHA})
444 for _, args := range steps {
445 cmd := exec.Command(toolpath.Look("git"), args...)
446 cmd.Env = append(os.Environ(), "GIT_SSH_COMMAND="+gitSSH, "GIT_TERMINAL_PROMPT=0")
447 cmd.Stdout, cmd.Stderr = sink, sink
448 if ok, why := runStep(cmd, deadline); !ok {
449 fmt.Fprintf(sink, "git %s: %s\n", args[0], why)
450 return &failure{Reason: "git " + args[0] + ": " + why}
451 }
452 }
453
454 env := stepEnv(j, home, r.buildSSH(j.SSH))
455 return r.runSteps(j, dir, env, sink, deadline, runStep)
456}
457
458// buildHome is a build's HOME and what to do with it when the build ends.
459//
460// Not the workspace, which is removed after every build: the Go module
461// cache and every other tool cache live under HOME. Not the runner's own
462// home either, where its SSH key and credential dotfiles are.
463//
464// A trusted build gets its repository's home,
465// <workdir>/trusted-home/<owner>/<name>, kept between builds so the
466// caches survive. One per repository: shared across repositories, a step
467// could poison a cache or plant a .gitconfig that another repository's
468// build would honour (#184). The root is not <workdir>/home, where homes
469// that untrusted builds could write were kept before #255, so none of
470// those is read again.
471//
472// An untrusted build gets <workdir>/build-<id>-home, new and empty,
473// removed when the build ends. The container mounts HOME read-write, so
474// a home a fork's build could write is a cache a stranger controls
475// (#255).
476func buildHome(workdir string, j job) (string, func(), error) {
477 if !j.Trusted {
478 dir := filepath.Join(workdir, fmt.Sprintf("build-%d-home", j.ID))
479 if err := os.Mkdir(dir, 0o700); err != nil {
480 return "", nil, err
481 }
482 return dir, func() {
483 if err := removeTree(dir); err != nil {
484 log.Printf("build %d: removing its home: %v", j.ID, err)
485 }
486 }, nil
487 }
488 root := filepath.Join(workdir, "trusted-home")
489 dir := filepath.Join(root, filepath.FromSlash(j.Repo))
490 if rel, err := filepath.Rel(root, dir); err != nil || rel == "." || strings.HasPrefix(rel, "..") {
491 return "", nil, fmt.Errorf("repository path %q escapes the build home root", j.Repo)
492 }
493 if err := os.MkdirAll(dir, 0o700); err != nil {
494 return "", nil, err
495 }
496 return dir, func() {}, nil
497}
498
499// removeTree deletes dir and everything under it. os.RemoveAll alone
500// fails on a directory without write permission, and the Go module cache
501// makes every directory it fills read-only.
502func removeTree(dir string) error {
503 filepath.WalkDir(dir, func(p string, d fs.DirEntry, err error) error {
504 if err == nil && d.IsDir() {
505 os.Chmod(p, 0o700)
506 }
507 return nil
508 })
509 return os.RemoveAll(dir)
510}
511
512// loopbackRemote reports whether the runner polls the daemon on its own
513// host over loopback.
514func (r *runner) loopbackRemote() bool {
515 _, host, ok := strings.Cut(r.remote, "@")
516 if !ok {
517 host = r.remote
518 }
519 return host == "127.0.0.1" || host == "localhost" || host == "::1"
520}
521
522// hostAddr is the address a podman build under pasta reaches the host
523// at: pasta's --map-guest-addr, which podman sets to this address and
524// which pasta translates to the host's public address.
525const hostAddr = "169.254.1.2"
526
527// buildSSH is the instance's ssh destination as a build reaches it.
528// Under pasta a podman build holds the host's own public address, so the
529// instance's public name resolves to the container itself. A runner
530// polling over loopback runs on the daemon's host, and its podman builds
531// get hostAddr with the user from -remote and the port from the claim's
532// destination when it is not 22. Any other runner's -remote is the path
533// that reaches the forge from where it runs, so its builds get that.
534func (r *runner) buildSSH(public string) string {
535 if r.isolation == isolationPodman && r.loopbackRemote() {
536 user, _, ok := strings.Cut(r.remote, "@")
537 if !ok || user == "" {
538 user = "git"
539 }
540 dest := hostAddr
541 _, hostport, ok := strings.Cut(public, "@")
542 if !ok {
543 hostport = public
544 }
545 if _, port, err := net.SplitHostPort(hostport); err == nil && port != "" && port != "22" {
546 dest = net.JoinHostPort(hostAddr, port)
547 }
548 return user + "@" + dest
549 }
550 return r.remote
551}
552
553// buildNetwork is the podman network option for a build on the daemon's
554// host. A build reaches the host at hostAddr, which pasta translates to
555// the host's public address, and cannot reach the host's loopback, so
556// none of its connections arrive from 127.0.0.1, the address the runner
557// polls from; the SSH auth limiter counts failures per source address
558// (#260). podman passes --no-map-gw to pasta by default; it is stated
559// here so the build's view of the host does not depend on that default.
560// The host's nftables table (deploy/gitbay-runner-egress.nft) limits
561// what a build reaches on the host to 22, 80 and 443.
562func (r *runner) buildNetwork() []string {
563 if !r.loopbackRemote() {
564 return nil
565 }
566 return []string{"--network", "pasta:--no-map-gw"}
567}
568
569// stepEnv builds the environment a build step runs with. It is
570// constructed, not inherited: os.Environ() would hand repository content
571// the runner's entire environment, including anything an operator set on
572// the service (#144).
573//
574// HOME is the build's home (buildHome), not the runner's own: tools read
575// credentials out of dotfiles — .netrc, .npmrc, .gitconfig — and a build
576// has no business finding the runner's.
577//
578// PATH is the one thing carried over: without it a step cannot find the
579// tools the host was provisioned with.
580func stepEnv(j job, home, sshDest string) []string {
581 path := os.Getenv("PATH")
582 if path == "" {
583 path = "/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin"
584 }
585 env := []string{
586 "PATH=" + path,
587 "HOME=" + home,
588 "LANG=C.UTF-8",
589 "CI=true",
590 "GITBAY_REPO=" + j.Repo,
591 "GITBAY_SHA=" + j.SHA,
592 "GITBAY_REF=" + j.Ref,
593 "GITBAY_JOB=" + j.Job,
594 "GITBAY_SSH=" + sshDest,
595 }
596 // The server sends secrets only for a trusted build. The claim's
597 // trust flag decides here as well, not whether any arrived (#255).
598 if j.Trusted {
599 for name, value := range j.Secrets {
600 env = append(env, name+"="+value)
601 }
602 }
603 return env
604}
605
606// ssh runs one control command against the server and returns stdout.// ssh runs one control command against the server and returns stdout.
607func (r *runner) ssh(stdin io.Reader, args ...string) (string, error) {
608 cmd := exec.Command(toolpath.Look("ssh"), append(append(r.sshOpts, r.remote), args...)...)
609 if stdin != nil {
610 cmd.Stdin = stdin
611 }
612 var out, errOut strings.Builder
613 cmd.Stdout, cmd.Stderr = &out, &errOut
614 if err := cmd.Run(); err != nil {
615 return out.String() + errOut.String(), err
616 }
617 return out.String(), nil
618}
619
620func fileExists(p string) bool { _, err := os.Stat(p); return err == nil }
621
622// defaultWorkdir picks a build workspace that another local user cannot
623// have created first.
624//
625// The default used to be <tmp>/gitbay-runner: a fixed name inside a
626// world-writable directory, created with MkdirAll, which succeeds against
627// an existing directory whoever owns it. On a shared host another user
628// could have made it — or symlinked it — before the runner started, and
629// this is the process that clones repositories and exports build secrets
630// into step environments (go:S5445, #153).
631//
632// The user's cache directory is not world-writable and is per-user by
633// construction. Falling back to tmp keeps a runner working where HOME is
634// unset, and checkWorkdir refuses the unsafe cases there.
635func defaultWorkdir() string {
636 if cache, err := os.UserCacheDir(); err == nil && cache != "" {
637 return filepath.Join(cache, "gitbay-runner")
638 }
639 return filepath.Join(os.TempDir(), "gitbay-runner")
640}
641
642// checkWorkdir makes sure the workspace is a directory this user owns
643// privately. MkdirAll is happy with one that already exists, so being
644// able to create it proves nothing about who made it.
645//
646// A directory we own that is merely too permissive is tightened rather
647// than refused: every runner before this one created its workspace 0755,
648// so refusing would take the runner down on upgrade to fix a permission
649// we are entitled to change. What cannot be repaired — a symlink, or
650// something owned by someone else — is refused, because those are what an
651// attacker leaves behind and neither is ours to correct.
652func checkWorkdir(dir string) error {
653 fi, err := os.Lstat(dir)
654 if err != nil {
655 return err
656 }
657 if fi.Mode()&os.ModeSymlink != 0 {
658 return fmt.Errorf("workdir %s is a symlink; point -workdir at a real directory", dir)
659 }
660 if !fi.IsDir() {
661 return fmt.Errorf("workdir %s is not a directory", dir)
662 }
663 if st, ok := fi.Sys().(*syscall.Stat_t); ok && int(st.Uid) != os.Getuid() {
664 return fmt.Errorf("workdir %s is owned by uid %d, not this process's %d", dir, st.Uid, os.Getuid())
665 }
666 if perm := fi.Mode().Perm(); perm&0o077 != 0 {
667 log.Printf("workdir %s was mode %04o; tightening to 0700 (builds and their secrets are this user's alone)", dir, perm)
668 if err := os.Chmod(dir, 0o700); err != nil {
669 return fmt.Errorf("tightening workdir %s: %w", dir, err)
670 }
671 }
672 return nil
673}