// Package symbols finds definitions in source files and keeps one index of // them per repository, for `repo symbols` and the blob view's links from a // name to where it is defined. // // The indexer is pure Go, with no external tagger. Go is parsed with // go/parser; the other languages are matched a line at a time against a // small table of anchored patterns, which finds the common definition // shapes and misses the unusual ones. // // Go function, method (named Type.Method), type, const, var // Swift function, class, struct, enum, interface (protocol), type (typealias) // Rust function, struct, enum, interface (trait), type, module, const, macro // Python function, method (an indented def), class // JavaScript, TS function, class, interface, type, enum, const // C and C++ headers function (prototype), struct, enum, class, type (typedef), macro (#define) // Shell function // Org, Markdown section (a heading) // // The index covers the tree of the default branch's head and is keyed by // that tree's id, so a push that leaves the tree as it was is not indexed // again. Worker builds it in the background after a push to the default // branch. package symbols import ( "bytes" "go/ast" "go/parser" "go/token" "path" "regexp" "strings" ) // Kinds are the symbol kinds the indexer records, in the order they are // listed. var Kinds = []string{ "function", "method", "class", "struct", "enum", "interface", "type", "const", "var", "module", "macro", "section", } // ValidKind reports whether k is one of Kinds. func ValidKind(k string) bool { for _, v := range Kinds { if v == k { return true } } return false } // Symbol is one definition in one file. Name is what is listed; Key is the // name as it is written where the symbol is used, which differs from Name // only for Go methods (Name "Server.Handle", Key "Handle"). type Symbol struct { Name string Key string Kind string Line int } // MaxFileBytes is the largest file the indexer reads. const MaxFileBytes = 1 << 20 // Skip reports whether a file is left out of the index before it is read: // too large, vendored, generated by its name, or in a language the indexer // does not know. func Skip(p string, size int64) bool { if size > MaxFileBytes || Language(p) == "" { return true } for _, seg := range strings.Split(path.Dir(p), "/") { if seg == "vendor" || seg == "node_modules" { return true } } base := path.Base(p) return strings.HasSuffix(base, "_gen.go") || strings.HasSuffix(base, ".pb.go") || strings.HasSuffix(base, ".min.js") } var languages = map[string]string{ ".go": "go", ".swift": "swift", ".rs": "rust", ".py": "python", ".js": "js", ".mjs": "js", ".cjs": "js", ".jsx": "js", ".ts": "js", ".tsx": "js", ".mts": "js", ".h": "c", ".hh": "c", ".hpp": "c", ".hxx": "c", ".sh": "shell", ".bash": "shell", ".zsh": "shell", ".org": "org", ".md": "markdown", ".markdown": "markdown", } // Language names the indexer's language for a path, or "" when it has // none. func Language(p string) string { return languages[strings.ToLower(path.Ext(p))] } // generatedGo is the marker `go generate` tools write, per the Go // convention for generated files. var generatedGo = regexp.MustCompile(`(?m)^// Code generated .* DO NOT EDIT\.$`) // MaxNameBytes is the longest name the index keeps. A longer one is // dropped: no real definition is named that way, and a file of them is a // way to fill the index. const MaxNameBytes = 256 // Extract returns the definitions in one file, in line order. func Extract(p string, data []byte) []Symbol { var syms []Symbol switch lang := Language(p); lang { case "": return nil case "go": syms = extractGo(data) case "org", "markdown": syms = extractHeadings(lang, data) default: syms = extractLines(rules[lang], data) } kept := syms[:0] for _, s := range syms { if len(s.Name) <= MaxNameBytes && len(s.Key) <= MaxNameBytes { kept = append(kept, s) } } return kept } func extractGo(data []byte) []Symbol { head := data if len(head) > 4096 { head = head[:4096] } if generatedGo.Match(head) { return nil } fset := token.NewFileSet() // A file that does not parse still yields the declarations before the // error. f, _ := parser.ParseFile(fset, "", data, parser.SkipObjectResolution) if f == nil { return nil } var out []Symbol add := func(id *ast.Ident, name, kind string) { if id == nil || id.Name == "_" { return } out = append(out, Symbol{Name: name, Key: id.Name, Kind: kind, Line: fset.Position(id.Pos()).Line}) } for _, d := range f.Decls { switch d := d.(type) { case *ast.FuncDecl: if d.Recv == nil || len(d.Recv.List) == 0 { add(d.Name, d.Name.Name, "function") continue } recv := receiverName(d.Recv.List[0].Type) if recv == "" { add(d.Name, d.Name.Name, "method") continue } add(d.Name, recv+"."+d.Name.Name, "method") case *ast.GenDecl: for _, spec := range d.Specs { switch s := spec.(type) { case *ast.TypeSpec: add(s.Name, s.Name.Name, "type") case *ast.ValueSpec: kind := "var" if d.Tok == token.CONST { kind = "const" } for _, n := range s.Names { add(n, n.Name, kind) } } } } } return out } // receiverName is the type a method is declared on, without a pointer or // type parameters. func receiverName(e ast.Expr) string { for { switch t := e.(type) { case *ast.StarExpr: e = t.X case *ast.ParenExpr: e = t.X case *ast.IndexExpr: e = t.X case *ast.IndexListExpr: e = t.X case *ast.Ident: return t.Name default: return "" } } } // rule is one line pattern: the first submatch is the name. type rule struct { re *regexp.Regexp kind string } func r(kind, pattern string) rule { return rule{regexp.MustCompile(pattern), kind} } // ident is a name in the languages the line rules cover. const ident = `([A-Za-z_$][A-Za-z0-9_$]*)` var rules = map[string][]rule{ "swift": { r("function", `^\s*(?:@\w+\s+)*(?:(?:public|private|fileprivate|internal|open|static|class|final|override|mutating|nonisolated|async|convenience|required)\s+)*func\s+`+ident), r("class", `^\s*(?:@\w+\s+)*(?:(?:public|private|fileprivate|internal|open|final)\s+)*(?:class|actor)\s+`+ident), r("struct", `^\s*(?:@\w+\s+)*(?:(?:public|private|fileprivate|internal)\s+)*struct\s+`+ident), r("enum", `^\s*(?:@\w+\s+)*(?:(?:public|private|fileprivate|internal|indirect)\s+)*enum\s+`+ident), r("interface", `^\s*(?:@\w+\s+)*(?:(?:public|private|fileprivate|internal)\s+)*protocol\s+`+ident), r("type", `^\s*(?:(?:public|private|fileprivate|internal)\s+)*typealias\s+`+ident), }, "rust": { r("function", `^\s*(?:pub(?:\([^)]*\))?\s+)?(?:(?:const|async|unsafe|extern(?:\s+"[^"]*")?)\s+)*fn\s+`+ident), r("struct", `^\s*(?:pub(?:\([^)]*\))?\s+)?struct\s+`+ident), r("enum", `^\s*(?:pub(?:\([^)]*\))?\s+)?enum\s+`+ident), r("interface", `^\s*(?:pub(?:\([^)]*\))?\s+)?(?:unsafe\s+)?trait\s+`+ident), r("type", `^\s*(?:pub(?:\([^)]*\))?\s+)?type\s+`+ident), r("module", `^\s*(?:pub(?:\([^)]*\))?\s+)?mod\s+`+ident), r("const", `^\s*(?:pub(?:\([^)]*\))?\s+)?(?:const|static(?:\s+mut)?)\s+`+ident+`\s*:`), r("macro", `^\s*macro_rules!\s+`+ident), }, "python": { r("function", `^(?:async\s+)?def\s+`+ident), r("method", `^\s+(?:async\s+)?def\s+`+ident), r("class", `^\s*class\s+`+ident), }, "js": { r("function", `^\s*(?:export\s+)?(?:default\s+)?(?:async\s+)?function\s*\*?\s*`+ident), r("class", `^\s*(?:export\s+)?(?:default\s+)?(?:abstract\s+)?class\s+`+ident), r("interface", `^\s*(?:export\s+)?(?:declare\s+)?interface\s+`+ident), r("type", `^\s*(?:export\s+)?(?:declare\s+)?type\s+`+ident+`\s*(?:<[^=]*>)?\s*=`), r("enum", `^\s*(?:export\s+)?(?:declare\s+)?(?:const\s+)?enum\s+`+ident), r("const", `^(?:export\s+)?const\s+`+ident+`\s*(?::[^=]*)?=`), }, "c": { r("macro", `^\s*#\s*define\s+`+ident), r("struct", `^\s*(?:typedef\s+)?struct\s+`+ident+`\s*\{`), r("enum", `^\s*(?:typedef\s+)?enum\s+(?:class\s+)?`+ident+`\s*(?::[^{]*)?\{`), r("class", `^\s*class\s+`+ident+`\s*(?::[^{;]*)?\{`), r("type", `^\s*typedef\s+[^;(]*?\b`+ident+`\s*;`), r("type", `^\s*}\s*`+ident+`\s*;`), r("function", `^[A-Za-z_][\w\s\*&:<>,]*?[\s\*&]`+ident+`\s*\([^;{]*\)\s*(?:const\s*)?;`), }, "shell": { r("function", `^\s*function\s+([A-Za-z_][A-Za-z0-9_:.-]*)`), r("function", `^\s*([A-Za-z_][A-Za-z0-9_:.-]*)\s*\(\)\s*(?:\{|$)`), }, } // keywords are never names, though a rule can capture one: Swift's // `class var x` reads as a class named var, C's `if (x);` as a prototype. var keywords = map[string]bool{ "return": true, "if": true, "while": true, "for": true, "switch": true, "sizeof": true, "else": true, "case": true, "do": true, "goto": true, "var": true, "let": true, "func": true, "static": true, } func extractLines(rs []rule, data []byte) []Symbol { var out []Symbol line := 0 for len(data) > 0 { line++ var text []byte if i := bytes.IndexByte(data, '\n'); i >= 0 { text, data = data[:i], data[i+1:] } else { text, data = data, nil } if len(text) > 1000 { continue } for _, ru := range rs { m := ru.re.FindSubmatch(text) if m == nil { continue } name := string(m[1]) if keywords[name] { continue } out = append(out, Symbol{Name: name, Key: name, Kind: ru.kind, Line: line}) break } } return out } var ( mdHeading = regexp.MustCompile(`^#{1,6}\s+(.+?)\s*#*\s*$`) orgHeading = regexp.MustCompile(`^\*+\s+(.+?)\s*$`) ) // extractHeadings lists headings as sections, skipping what sits inside a // code block, where a line starting with # or * is code. func extractHeadings(lang string, data []byte) []Symbol { var out []Symbol inBlock := false for i, text := range strings.Split(string(data), "\n") { trimmed := strings.TrimSpace(text) if lang == "markdown" { if strings.HasPrefix(trimmed, "```") || strings.HasPrefix(trimmed, "~~~") { inBlock = !inBlock continue } } else { lower := strings.ToLower(trimmed) if strings.HasPrefix(lower, "#+begin_") { inBlock = true continue } if strings.HasPrefix(lower, "#+end_") { inBlock = false continue } } if inBlock || len(text) > 1000 { continue } re := mdHeading if lang == "org" { re = orgHeading } if m := re.FindStringSubmatch(text); m != nil { out = append(out, Symbol{Name: m[1], Key: m[1], Kind: "section", Line: i + 1}) } } return out }