mirror of
https://github.com/anchore/syft.git
synced 2026-08-19 08:38:25 +02:00
* fix(elf): bound compressed ELF section reads `debug/elf` takes a section's decompressed size from that section's own compression header, and a highly compressible stream really does deliver the bytes that header promises, so `internal/saferio` does not help: it faithfully allocates every one of them. A 2MB input file drives `elf.NewFile` to allocate over 10GB and return no error, which is a fatal OOM rather than a recoverable panic. Which sections get read is not up to the caller. `elf.NewFile` always reads the section-name string table, and `File.Symbols` reads `.symtab` plus whatever section its `Link` field points at, so being selective about sections is not enough to avoid it. New `elfutil.NewFile` is a drop-in for `elf.NewFile` that rejects a declared decompressed size over 128MB. Every production call site goes through it, and a ruleguard rule keeps the next one from going direct. The check runs in two parts, since `debug/elf` expands sections at two different times. The section-name string table is the only one `elf.NewFile` expands itself, so it is checked against the raw bytes before the call; everything else is expanded lazily by `(*Section).Open` and is checked after the parse, where names, types and decompressed sizes are already resolved. Only the sections syft can actually reach are bounded, which keeps the guard from costing real binaries. DWARF is excluded since nothing calls `File.DWARF`, so a large compressed `.debug_info` no longer skips the whole file, and sections `debug/elf` will not decompress anyway (`SHF_ALLOC`, `SHT_NOBITS`) are left alone. The legacy `.zdebug` form is matched on the section name the way `debug/elf` gates it rather than on the `ZLIB` magic, so an ordinary section starting with those four bytes is not mistaken for a compressed one. Signed-off-by: Alex Goodman <wagoodman@users.noreply.github.com> * fix(elf): gate debug/buildinfo behind the compressed-section check `debug/buildinfo.Read` opens ELF files with `debug/elf` itself, and `elf.NewFile` expands the section-name string table as it parses, so the golang cataloger was still reachable by the same bomb `elfutil` exists to stop. A 261KB fixture drove 1.4GB of allocation through `buildinfo.Read` and returned no error. `elfutil.CheckSectionNameTable` is now exported for that case: callers that cannot use `NewFile` because the `debug/elf` call is made for them inside another package. Both `buildinfo.Read` call sites go through it, including the UPX-decompressed one. Also corrects claims that did not hold up: - the package doc's 2MB-to-10GB figure is not reachable with zlib (~1000:1), so it now carries the measured 510KB-to-2.6GB, and names zstd's 32767:1 since that is what makes the small inputs possible - `.go.buildinfo` was listed as a hot-path section elfutil covers, but it is read through `debug/buildinfo` and never touches `Section.Data` - the graalvm comment claimed routing size rejections away from `*elf.FormatError` improved reporting; both branches are skipped by the caller and only the FormatError branch logs, so it did the opposite - `sharedLibraries` logged short and truncated files as real ELF failures, since `debug/elf` returns a bare `io.EOF` rather than an `*elf.FormatError` for those Signed-off-by: Alex Goodman <wagoodman@users.noreply.github.com> * added test comments around the negative cases Signed-off-by: Alex Goodman <wagoodman@users.noreply.github.com> * additional tests Signed-off-by: Alex Goodman <wagoodman@users.noreply.github.com> * better decomposition and comments Signed-off-by: Alex Goodman <wagoodman@users.noreply.github.com> * use a less brittle constant for error detection Signed-off-by: Alex Goodman <wagoodman@users.noreply.github.com> --------- Signed-off-by: Alex Goodman <wagoodman@users.noreply.github.com>
444 lines
16 KiB
Go
444 lines
16 KiB
Go
package golang
|
|
|
|
import (
|
|
"bytes"
|
|
"debug/gosym"
|
|
"debug/macho"
|
|
"encoding/binary"
|
|
"fmt"
|
|
"io"
|
|
"runtime/debug"
|
|
"slices"
|
|
"strings"
|
|
|
|
"github.com/anchore/syft/syft/internal/elfutil"
|
|
)
|
|
|
|
// mainPackage is the import path the linker assigns to the binary's main package.
|
|
const mainPackage = "main"
|
|
|
|
// vendorPrefix is the import-path prefix carried by vendored packages (e.g. from `go mod vendor`, or the
|
|
// standard library's own vendored dependencies such as "vendor/golang.org/x/net/http2").
|
|
const vendorPrefix = "vendor/"
|
|
|
|
// binarySymbol represents a single function symbol extracted from a go binary's pclntab.
|
|
type binarySymbol struct {
|
|
// packagePath is the import path of the package that owns the symbol (e.g. "github.com/foo/bar/internal/baz")
|
|
packagePath string
|
|
|
|
// name is the fully qualified symbol name (e.g. "github.com/foo/bar/internal/baz.(*Type).Method")
|
|
name string
|
|
}
|
|
|
|
// getSymbols extracts all function symbols from the pclntab of a go binary. The pclntab is required by the
|
|
// go runtime (for panic tracebacks and GC), so it is present even in binaries built with -ldflags="-s -w".
|
|
func getSymbols(r io.ReaderAt) (syms []binarySymbol, err error) {
|
|
defer func() {
|
|
if r := recover(); r != nil {
|
|
// the gosym package can panic on malformed pclntab data
|
|
syms = nil
|
|
err = fmt.Errorf("recovered from panic while reading pclntab: %v", r)
|
|
}
|
|
}()
|
|
|
|
pclntab, textStart, err := readPclntab(r)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
|
|
table, err := gosym.NewTable(nil, gosym.NewLineTable(pclntab, textStart))
|
|
if err != nil {
|
|
return nil, fmt.Errorf("unable to parse pclntab: %w", err)
|
|
}
|
|
|
|
seen := make(map[string]struct{})
|
|
for _, fn := range table.Funcs {
|
|
if fn.Sym == nil || isCompilerGeneratedName(fn.Name) {
|
|
continue
|
|
}
|
|
seen[fn.Name] = struct{}{}
|
|
syms = append(syms, makeBinarySymbol(fn.Name, fn.PackageName()))
|
|
}
|
|
|
|
// debug/gosym only exposes top-level functions; functions that the compiler inlined into their
|
|
// callers are absent from table.Funcs even though their names are recorded in the pclntab funcname
|
|
// table (used to reconstruct inlined frames in tracebacks). Recover those names so that a
|
|
// vulnerable-but-inlined function (e.g. a small stdlib wrapper) is still reported as present.
|
|
for _, name := range funcNameTable(pclntab) {
|
|
if _, ok := seen[name]; ok {
|
|
continue
|
|
}
|
|
pkgPath := packagePathFromSymbolName(name)
|
|
if pkgPath == "" {
|
|
continue
|
|
}
|
|
seen[name] = struct{}{}
|
|
syms = append(syms, makeBinarySymbol(name, pkgPath))
|
|
}
|
|
|
|
return syms, nil
|
|
}
|
|
|
|
// packagePathFromSymbolName derives the owning package import path from a fully qualified symbol name.
|
|
// The package path is everything up to the first "." that follows the final "/" — e.g.
|
|
// "path/filepath.IsLocal" -> "path/filepath" and "golang.org/x/net/html.(*Tokenizer).Next" ->
|
|
// "golang.org/x/net/html". Returns "" when the name has no package-qualifying dot or is compiler-generated.
|
|
func packagePathFromSymbolName(name string) string {
|
|
if isCompilerGeneratedName(name) {
|
|
return ""
|
|
}
|
|
name = nameWithoutTypeArgs(name)
|
|
slash := strings.LastIndex(name, "/")
|
|
dot := strings.IndexByte(name[slash+1:], '.')
|
|
if dot < 0 {
|
|
return ""
|
|
}
|
|
return name[:slash+1+dot]
|
|
}
|
|
|
|
// makeBinarySymbol builds a binarySymbol, unescaping the %xx sequences the go linker introduces in the
|
|
// import-path portion of a symbol name (see unescapePackagePath). The local-symbol suffix (method and
|
|
// function names) is left untouched, so only the path prefix shared by packagePath and name is rewritten.
|
|
func makeBinarySymbol(name, pkgPath string) binarySymbol {
|
|
unescaped := unescapePackagePath(pkgPath)
|
|
if unescaped != pkgPath && strings.HasPrefix(name, pkgPath) {
|
|
name = unescaped + name[len(pkgPath):]
|
|
}
|
|
return binarySymbol{packagePath: unescaped, name: name}
|
|
}
|
|
|
|
// unescapePackagePath reverses the escaping cmd/internal/objabi.PathToPrefix applies to import paths in
|
|
// symbol names: bytes like '.' (at or after the final '/'), '%', '"', control bytes, and high bytes are
|
|
// written as lowercase "%xx". For example "gopkg.in/yaml.v2" is stored as "gopkg.in/yaml%2ev2", so this
|
|
// restores it before matching against the (unescaped) module paths from build info. A lone or malformed
|
|
// '%' sequence is left as-is.
|
|
func unescapePackagePath(path string) string {
|
|
if !strings.Contains(path, "%") {
|
|
return path
|
|
}
|
|
var b strings.Builder
|
|
b.Grow(len(path))
|
|
for i := 0; i < len(path); i++ {
|
|
if path[i] == '%' && i+2 < len(path) {
|
|
if hi, ok1 := unhex(path[i+1]); ok1 {
|
|
if lo, ok2 := unhex(path[i+2]); ok2 {
|
|
b.WriteByte(hi<<4 | lo)
|
|
i += 2
|
|
continue
|
|
}
|
|
}
|
|
}
|
|
b.WriteByte(path[i])
|
|
}
|
|
return b.String()
|
|
}
|
|
|
|
func unhex(c byte) (byte, bool) {
|
|
switch {
|
|
case c >= '0' && c <= '9':
|
|
return c - '0', true
|
|
case c >= 'a' && c <= 'f':
|
|
return c - 'a' + 10, true
|
|
case c >= 'A' && c <= 'F':
|
|
return c - 'A' + 10, true
|
|
}
|
|
return 0, false
|
|
}
|
|
|
|
// nameWithoutTypeArgs strips the type-argument portion from an instantiated generic symbol name, e.g.
|
|
// "foo/bar.Do[net/url.Values]" -> "foo/bar.Do". The slashes and dots inside the brackets would otherwise
|
|
// corrupt package-path derivation (yielding "foo/bar.Do[net" for the example above). Mirrors
|
|
// debug/gosym's (*Sym).nameWithoutInst.
|
|
func nameWithoutTypeArgs(name string) string {
|
|
start := strings.IndexByte(name, '[')
|
|
if start < 0 {
|
|
return name
|
|
}
|
|
end := strings.LastIndexByte(name, ']')
|
|
if end < 0 {
|
|
// malformed: an opening bracket should always have a closing one
|
|
return name
|
|
}
|
|
return name[:start] + name[end+1:]
|
|
}
|
|
|
|
// oldStyleCompilerGeneratedPrefixes match compiler/linker-generated symbols from toolchains older than
|
|
// go1.20, which used "." where newer toolchains use ":" (e.g. "go.buildid" is now "go:buildid",
|
|
// "go.type.*" is now "go:type.*"). These must be prefix (not substring) matches, and a bare "go." prefix
|
|
// is not enough: legitimate module paths such as "go.uber.org/zap" also start with "go.".
|
|
var oldStyleCompilerGeneratedPrefixes = []string{
|
|
"go.buildid",
|
|
"go.builtin.",
|
|
"go.constinfo.",
|
|
"go.cuinfo.",
|
|
"go.func.",
|
|
"go.importpath.",
|
|
"go.info.",
|
|
"go.interface.",
|
|
"go.itab.",
|
|
"go.itablink.",
|
|
"go.map.",
|
|
"go.shape.",
|
|
"go.string.",
|
|
"go.type.",
|
|
"go.typelink.",
|
|
"type.",
|
|
}
|
|
|
|
// isCompilerGeneratedName reports whether a symbol name was synthesized by the compiler or linker rather
|
|
// than declared in Go source. Since go1.20 these names contain ':' or '..' (e.g. "type:.eq.*",
|
|
// "go:string.*") — byte sequences that never appear in a real Go import path or identifier. Older
|
|
// toolchains used '.' as the separator (e.g. "go.type.*", "type..hash.*"), which is matched against the
|
|
// known reserved prefixes. Such names belong to no package and are dropped rather than mis-attributed
|
|
// (e.g. bucketed under a bogus "type" stdlib package).
|
|
func isCompilerGeneratedName(name string) bool {
|
|
if strings.Contains(name, ":") || strings.Contains(name, "..") {
|
|
return true
|
|
}
|
|
for _, prefix := range oldStyleCompilerGeneratedPrefixes {
|
|
if strings.HasPrefix(name, prefix) {
|
|
return true
|
|
}
|
|
}
|
|
return false
|
|
}
|
|
|
|
// funcNameTable returns every function name recorded in the pclntab's funcname table, including the
|
|
// names of inlined functions that debug/gosym does not expose. It parses the pclntab header for the
|
|
// Go 1.16+ layouts; on any unrecognized layout or out-of-bounds offset it returns nil (fail-soft), so
|
|
// callers fall back to the debug/gosym function set. See the runtime's pcHeader / moduledata layout.
|
|
func funcNameTable(pclntab []byte) []string {
|
|
if len(pclntab) < 8 {
|
|
return nil
|
|
}
|
|
|
|
magic := binary.LittleEndian.Uint32(pclntab[0:4])
|
|
// the field before funcnameOffset is textStart, which exists in the 1.18+ headers but not 1.16/1.17
|
|
var hasTextStart bool
|
|
switch magic {
|
|
case 0xfffffff1, 0xfffffff0: // go1.20+, go1.18/1.19
|
|
hasTextStart = true
|
|
case 0xfffffffa: // go1.16/1.17
|
|
hasTextStart = false
|
|
default:
|
|
return nil
|
|
}
|
|
|
|
ptrSize := int(pclntab[7])
|
|
if ptrSize != 4 && ptrSize != 8 {
|
|
return nil
|
|
}
|
|
|
|
readWord := func(idx int) (uint64, bool) {
|
|
off := 8 + idx*ptrSize
|
|
if off+ptrSize > len(pclntab) {
|
|
return 0, false
|
|
}
|
|
if ptrSize == 8 {
|
|
return binary.LittleEndian.Uint64(pclntab[off : off+8]), true
|
|
}
|
|
return uint64(binary.LittleEndian.Uint32(pclntab[off : off+4])), true
|
|
}
|
|
|
|
// header words after (nfunc, nfiles): [textStart,] funcnameOffset, cuOffset, ...
|
|
funcnameIdx := 2
|
|
if hasTextStart {
|
|
funcnameIdx = 3
|
|
}
|
|
funcnameOffset, ok1 := readWord(funcnameIdx)
|
|
cuOffset, ok2 := readWord(funcnameIdx + 1)
|
|
if !ok1 || !ok2 {
|
|
return nil
|
|
}
|
|
|
|
start, end := int(funcnameOffset), int(cuOffset)
|
|
if start < 0 || end > len(pclntab) || start >= end {
|
|
return nil
|
|
}
|
|
|
|
var names []string
|
|
for raw := range bytes.SplitSeq(pclntab[start:end], []byte{0}) {
|
|
if len(raw) == 0 {
|
|
continue
|
|
}
|
|
names = append(names, string(raw))
|
|
}
|
|
return names
|
|
}
|
|
|
|
// readPclntab locates the pclntab and the start address of the text segment within the binary.
|
|
func readPclntab(r io.ReaderAt) (pclntab []byte, textStart uint64, err error) {
|
|
ident := make([]byte, 16)
|
|
if n, err := r.ReadAt(ident, 0); n < len(ident) || err != nil {
|
|
return nil, 0, errUnrecognizedFormat
|
|
}
|
|
|
|
switch {
|
|
case strings.HasPrefix(string(ident), "\x7FELF"):
|
|
f, err := elfutil.NewFile(r)
|
|
if err != nil {
|
|
return nil, 0, fmt.Errorf("unable to parse ELF binary: %w", err)
|
|
}
|
|
sect := f.Section(".gopclntab")
|
|
if sect == nil {
|
|
return nil, 0, fmt.Errorf("no .gopclntab section found")
|
|
}
|
|
pclntab, err := sect.Data()
|
|
if err != nil {
|
|
return nil, 0, fmt.Errorf("unable to read .gopclntab section: %w", err)
|
|
}
|
|
text := f.Section(".text")
|
|
if text == nil {
|
|
return nil, 0, fmt.Errorf("no .text section found")
|
|
}
|
|
return pclntab, text.Addr, nil
|
|
case strings.HasPrefix(string(ident), "\xFE\xED\xFA") || strings.HasPrefix(string(ident[1:]), "\xFA\xED\xFE"):
|
|
f, err := macho.NewFile(r)
|
|
if err != nil {
|
|
return nil, 0, fmt.Errorf("unable to parse Mach-O binary: %w", err)
|
|
}
|
|
sect := f.Section("__gopclntab")
|
|
if sect == nil {
|
|
return nil, 0, fmt.Errorf("no __gopclntab section found")
|
|
}
|
|
pclntab, err := sect.Data()
|
|
if err != nil {
|
|
return nil, 0, fmt.Errorf("unable to read __gopclntab section: %w", err)
|
|
}
|
|
text := f.Section("__text")
|
|
if text == nil {
|
|
return nil, 0, fmt.Errorf("no __text section found")
|
|
}
|
|
return pclntab, text.Addr, nil
|
|
}
|
|
|
|
// note: PE and XCOFF binaries do not place the pclntab in a dedicated section; locating it requires
|
|
// walking the symbol table for runtime.pclntab markers, which is not yet supported here
|
|
return nil, 0, errUnrecognizedFormat
|
|
}
|
|
|
|
// moduleSymbols attributes each extracted symbol to the module that owns it (by longest module path prefix
|
|
// of the symbol's package path) and returns, per module path, the symbols grouped by the import path of the
|
|
// owning package. Each inner value is a sorted, deduplicated list of symbol names local to that package
|
|
// (the import path prefix stripped, e.g. "github.com/foo/bar.(*T).M" under key "github.com/foo/bar" becomes
|
|
// "(*T).M"). Symbols from the "main" package are attributed to the main module and keyed by the "main"
|
|
// import path the linker assigns. Standard-library symbols (which belong to no module) are collected
|
|
// separately and returned as the second value, grouped by import path, so they can be attached to the
|
|
// synthetic "stdlib" package. Vendored packages carry a "vendor/" import-path prefix: such symbols match
|
|
// both modules whose own path carries the prefix and modules without it (matched with the prefix trimmed),
|
|
// and the prefix is retained in the group key only when the owning module itself is named "vendor/...".
|
|
// Module-less vendored packages (the stdlib's own vendored dependencies, e.g.
|
|
// "vendor/golang.org/x/net/http2") are dropped: stdlib vulnerabilities seem to be reported against the public
|
|
// packages (e.g. "crypto/x509"), not the vendored internal copies. Compiler/runtime-internal symbols that
|
|
// are neither module-owned nor a recognizable stdlib import path are likewise dropped.
|
|
func moduleSymbols(symbols []binarySymbol, main *debug.Module, deps []*debug.Module) (byModule map[string]map[string][]string, stdlib map[string][]string) {
|
|
if len(symbols) == 0 {
|
|
return nil, nil
|
|
}
|
|
|
|
var modulePaths []string
|
|
if main != nil && main.Path != "" {
|
|
modulePaths = append(modulePaths, main.Path)
|
|
}
|
|
for _, dep := range deps {
|
|
if dep != nil && dep.Path != "" {
|
|
modulePaths = append(modulePaths, dep.Path)
|
|
}
|
|
}
|
|
|
|
results := make(map[string]map[string][]string)
|
|
stdlib = make(map[string][]string)
|
|
for _, sym := range symbols {
|
|
importPath := sym.packagePath
|
|
|
|
// the linker renames the main package's import path to "main"; attribute it to the main module,
|
|
// but keep "main" as the group key since the original import path is not recoverable.
|
|
attrPath := importPath
|
|
if importPath == mainPackage && main != nil {
|
|
attrPath = main.Path
|
|
}
|
|
|
|
best := findBestMatch(modulePaths, attrPath)
|
|
|
|
// the vendor/ prefix is only retained when the owning module itself is named "vendor/...";
|
|
// in all other cases (non-vendored modules and vendored stdlib) the recorded import path is trimmed
|
|
if !strings.HasPrefix(best, vendorPrefix) {
|
|
importPath = strings.TrimPrefix(importPath, vendorPrefix)
|
|
}
|
|
|
|
local := localSymbolName(sym.name, importPath)
|
|
if best == "" {
|
|
if importPath != mainPackage && isStandardImportPath(importPath) { // drop stdlib vendored packages
|
|
stdlib[importPath] = append(stdlib[importPath], local)
|
|
}
|
|
continue
|
|
}
|
|
if results[best] == nil {
|
|
results[best] = make(map[string][]string)
|
|
}
|
|
results[best][importPath] = append(results[best][importPath], local)
|
|
}
|
|
|
|
for _, byImport := range results {
|
|
sortCompactGroups(byImport)
|
|
}
|
|
sortCompactGroups(stdlib)
|
|
if len(stdlib) == 0 {
|
|
stdlib = nil
|
|
}
|
|
|
|
return results, stdlib
|
|
}
|
|
|
|
// findBestMatch returns the module path that owns the given package path taking into account vendor/ prefixes: a vendor/ import path
|
|
// will take precedence and continue to match, non-vendored imports will match against their vendored equivalent
|
|
func findBestMatch(modulePaths []string, importPath string) string {
|
|
trimmedPath, trimmed := strings.CutPrefix(importPath, vendorPrefix)
|
|
candidatePaths := []string{importPath, trimmedPath}
|
|
if !trimmed {
|
|
candidatePaths = candidatePaths[:1]
|
|
}
|
|
|
|
var best string
|
|
for _, candidate := range candidatePaths {
|
|
for _, modPath := range modulePaths {
|
|
// the prefix must end at a path-segment boundary in candidate, so that e.g. the module
|
|
// "github.com/foo/bar" matches the package "github.com/foo/bar/baz" but not "github.com/foo/barbaz"
|
|
if len(modPath) > len(best) && strings.HasPrefix(candidate, modPath) && (candidate == modPath || candidate[len(modPath)] == '/') {
|
|
best = modPath
|
|
}
|
|
}
|
|
}
|
|
return best
|
|
}
|
|
|
|
// localSymbolName strips the owning package's import path prefix from a fully qualified symbol name, e.g.
|
|
// "github.com/foo/bar.(*T).M" with import path "github.com/foo/bar" becomes "(*T).M". The name is returned
|
|
// unchanged when it does not carry the expected prefix.
|
|
func localSymbolName(name, importPath string) string {
|
|
if !strings.HasPrefix(importPath, vendorPrefix) {
|
|
name = strings.TrimPrefix(name, vendorPrefix)
|
|
}
|
|
if len(importPath) < len(name) && strings.HasPrefix(name, importPath) && name[len(importPath)] == '.' {
|
|
return name[len(importPath)+1:]
|
|
}
|
|
return name
|
|
}
|
|
|
|
// sortCompactGroups sorts and deduplicates each symbol list in a group keyed by import path, in place.
|
|
func sortCompactGroups(groups map[string][]string) {
|
|
for path, names := range groups {
|
|
slices.Sort(names)
|
|
groups[path] = slices.Compact(names)
|
|
}
|
|
}
|
|
|
|
// isStandardImportPath reports whether path is a Go standard-library import path. This mirrors the rule
|
|
// the Go toolchain uses: a path is standard if the element before its first slash contains no dot (e.g.
|
|
// "net/http", "runtime", "internal/abi"), which distinguishes it from module paths like
|
|
// "github.com/foo/bar" whose leading element is a domain name.
|
|
func isStandardImportPath(path string) bool {
|
|
first, _, _ := strings.Cut(path, "/")
|
|
return first != "" && !strings.Contains(first, ".")
|
|
}
|