Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
18 changes: 13 additions & 5 deletions containers/l7_self_metrics.go
Original file line number Diff line number Diff line change
Expand Up @@ -12,14 +12,16 @@ import (
)

var (
// HPACKDecodeErrorsTotal counts HPACK decode failures in the HTTP/2
// parser: the decoder's dynamic table no longer matches the peer's, as
// when the agent joined a long-lived connection mid-stream or missed a
// HEADERS frame.
// HPACKDecodeErrorsTotal counts HTTP/2 header blocks that were not valid
// HPACK, or decoded to pseudo-headers that cannot be right for their
// direction: the decoder's dynamic table had drifted from the peer's, as
// when a HEADERS frame was lost unnoticed. References to entries inserted
// before the agent joined the connection are not errors; they are counted
// as the hpack_partial stage of node_agent_http2_stage_total.
HPACKDecodeErrorsTotal = prometheus.NewCounter(
prometheus.CounterOpts{
Name: "node_agent_hpack_decode_errors_total",
Help: "Total HPACK decode errors in HTTP/2 parser (mid-stream join indicator)",
Help: "HTTP/2 header blocks that failed to decode or decoded to implausible headers; the decoder's table is reset",
},
)

Expand Down Expand Up @@ -125,9 +127,15 @@ var (
// inferred. Stages, in order:
//
// stream_created client HEADERS decoded, request object created
// stream_evicted a request still waiting for its response dropped to
// make room: the connection had too many such requests,
// nearly always ones whose response was lost
// response_status :status seen on the response
// end_stream END_STREAM flag seen (a frame flag, not HPACK)
// completed both of the above -> request emitted
// hpack_partial a header block referenced table entries the decoder
// does not hold (inserted before it joined or was
// reset); the other headers in it were decoded
// hpack_error HPACK block failed to decode; decoder reset
//
// A request is only emitted with BOTH response_status and end_stream, so
Expand Down
20 changes: 10 additions & 10 deletions ebpftracer/ebpf.go

Large diffs are not rendered by default.

133 changes: 93 additions & 40 deletions ebpftracer/ebpf/l7/l7.c
Original file line number Diff line number Diff line change
Expand Up @@ -34,20 +34,33 @@
asm volatile ("%0 &= %1" : "+r"(size) : "i"(MAX_PAYLOAD_SIZE-1)); \
})

// TRUNCATE_COPY_SIZE bounds a copy into a MAX_PAYLOAD_SIZE buffer (an
// event's payload or response, a request's payload) to the whole buffer.
// TRUNCATE_PAYLOAD_SIZE stops one byte short, while payload_size and
// userspace allow MAX_PAYLOAD_SIZE: a 4096-byte write, which is how Go's
// HTTP/2 flushes its 4 KB write buffer, was decoded with a last byte the
// event never wrote, inside whichever frame crossed the end of the buffer.
//
// The clamp is asm so the verifier sees the compare on the register the copy
// uses: written in C, clang compared a copy and the verifier lost the bound.
#define TRUNCATE_COPY_SIZE(size) ({ \
asm volatile ("if %0 <= %1 goto +1\n\t%0 = %1" \
: "+r"(size) : "i"(MAX_PAYLOAD_SIZE)); \
})

// COPY_PAYLOAD for use with non-ringbuf allocations (l7_request heap)
#define COPY_PAYLOAD(dst, size, src) ({ \
TRUNCATE_PAYLOAD_SIZE(size); \
TRUNCATE_COPY_SIZE(size); \
if (bpf_probe_read(dst, size, src)) { \
return 0; \
} \
})

// COPY_PAYLOAD_RINGBUF for use with ring buffer allocations
// Discards the event and returns 0 on failure to satisfy eBPF verifier
// COPY_PAYLOAD_RINGBUF copies a payload into an event from reserve_l7_event
// and returns 0 (dropping the event) if the read fails.
#define COPY_PAYLOAD_RINGBUF(e, dst, size, src) ({ \
TRUNCATE_PAYLOAD_SIZE(size); \
TRUNCATE_COPY_SIZE(size); \
if (bpf_probe_read(dst, size, src)) { \
bpf_ringbuf_discard(e, 0); \
return 0; \
} \
})
Expand Down Expand Up @@ -217,8 +230,34 @@ struct user_msghdr {
__u32 msg_flags;
};

// send_event submits an L7 event to the ring buffer
// The event must have been allocated via bpf_ringbuf_reserve(&l7_events, ...)
// l7_ringbuf_drops counts L7 events lost because l7_events was full.
// Userspace exports it as node_agent_l7_ringbuf_drops_total.
struct {
__uint(type, BPF_MAP_TYPE_PERCPU_ARRAY);
__uint(key_size, sizeof(__u32));
__uint(value_size, sizeof(__u64));
__uint(max_entries, 1);
} l7_ringbuf_drops SEC(".maps");

// L7_EVENT_LEN_MASK is the smallest all-ones mask covering sizeof(struct
// l7_event); send_event uses it to bound the record length for the verifier.
#define L7_EVENT_LEN_MASK 0x3fff

// l7_event_heap is where an event is built before it is copied into
// l7_events at its actual size (see send_event).
struct {
__uint(type, BPF_MAP_TYPE_PERCPU_ARRAY);
__type(key, int);
__type(value, struct l7_event);
__uint(max_entries, 1);
} l7_event_heap SEC(".maps");

// send_event copies an L7 event built by reserve_l7_event into the ring
// buffer, sized to what it carries. Records used to be reserved at the full
// struct size, two MAX_PAYLOAD_SIZE buffers whatever the payload, so a burst
// of small events (HTTP/2 frames of a streaming response) filled the buffer
// with mostly empty slots and later events were lost. Payload and response
// stay at their fixed offsets, so userspace decodes either size the same way.
static inline __attribute__((__always_inline__))
void send_event(void *ctx, struct l7_event *e, struct connection_id cid, struct connection *conn) {
e->connection_timestamp = conn->timestamp;
Expand All @@ -243,46 +282,49 @@ void send_event(void *ctx, struct l7_event *e, struct connection_id cid, struct
e->socket_info_valid = 0;
}

bpf_ringbuf_submit(e, 0);
}

// l7_ringbuf_drops counts L7 events lost because l7_events was full.
// Userspace exports it as node_agent_l7_ringbuf_drops_total.
struct {
__uint(type, BPF_MAP_TYPE_PERCPU_ARRAY);
__uint(key_size, sizeof(__u32));
__uint(value_size, sizeof(__u64));
__uint(max_entries, 1);
} l7_ringbuf_drops SEC(".maps");

// reserve_l7_event allocates an l7_event from the ring buffer
// Returns NULL if the ring buffer is full (backpressure)
static inline __attribute__((__always_inline__))
struct l7_event *reserve_l7_event(void) {
struct l7_event *e = bpf_ringbuf_reserve(&l7_events, sizeof(struct l7_event), 0);
if (!e) {
__u64 data = e->response_size ? e->response_size : e->payload_size;
if (data > MAX_PAYLOAD_SIZE) {
data = MAX_PAYLOAD_SIZE;
}
__u64 len = (e->response_size ? __builtin_offsetof(struct l7_event, response)
: __builtin_offsetof(struct l7_event, payload)) + data;
// Bound len for the verifier: the mask makes it non-negative and small,
// the comparison caps it at the event's size.
asm volatile ("%0 &= %1" : "+r"(len) : "i"(L7_EVENT_LEN_MASK));
if (len > sizeof(struct l7_event)) {
len = sizeof(struct l7_event);
}
if (bpf_ringbuf_output(&l7_events, e, len, 0)) {
__u32 zero = 0;
__u64 *drops = bpf_map_lookup_elem(&l7_ringbuf_drops, &zero);
if (drops) {
*drops += 1;
}
} else {
// Initialize event to zero state
e->protocol = PROTOCOL_UNKNOWN;
e->status = STATUS_UNKNOWN;
e->method = METHOD_UNKNOWN;
e->statement_id = 0;
e->payload_size = 0;
e->response_size = 0;
}
}

// reserve_l7_event returns this CPU's scratch event, with its header reset.
// Nothing is taken from the ring buffer until send_event, so an event that is
// built and then dropped (most reads turn out not to need one) costs no ring
// space.
static inline __attribute__((__always_inline__))
struct l7_event *reserve_l7_event(void) {
int zero = 0;
struct l7_event *e = bpf_map_lookup_elem(&l7_event_heap, &zero);
if (!e) {
return 0;
}
__builtin_memset(e, 0, __builtin_offsetof(struct l7_event, payload));
Comment thread
mayankpande88 marked this conversation as resolved.
e->protocol = PROTOCOL_UNKNOWN;
e->status = STATUS_UNKNOWN;
e->method = METHOD_UNKNOWN;
return e;
}

// discard_l7_event discards a reserved event without sending
// Use when an error occurs after reserve but before send
// discard_l7_event drops an event from reserve_l7_event without sending it.
// The scratch buffer needs no release; this marks where an event is dropped.
static inline __attribute__((__always_inline__))
void discard_l7_event(struct l7_event *e) {
bpf_ringbuf_discard(e, 0);
}

static inline __attribute__((__always_inline__))
Expand All @@ -302,8 +344,8 @@ __u64 read_iovec(char *iovec, __u64 iovlen, __u64 ret, char *buf, __u64 *total_s
}

*total_size = iov.size;
__u64 size = MIN(iov.size, MAX_PAYLOAD_SIZE);
TRUNCATE_PAYLOAD_SIZE(size);
__u64 size = iov.size;
TRUNCATE_COPY_SIZE(size);

// Direct copy without offset arithmetic on map values
if (bpf_probe_read(buf, size, (void *)iov.buf)) {
Expand Down Expand Up @@ -596,8 +638,19 @@ int trace_enter_write(void *ctx, __u64 fd, __u16 is_tls, char *buf, __u64 size,

// Port-based HTTP/2 hint: Try HTTP/2 detection first for HTTPS traffic (port 443/8443)
// Most modern HTTPS traffic uses HTTP/2, and this helps detect gRPC DATA frames
// that don't have the connection preface
if (conn->dport != 53 && http2_detection_allowed(conn) && is_likely_http2_port(conn->dport) && looks_like_http2_frame(payload, size, METHOD_HTTP2_CLIENT_FRAMES)) {
// that don't have the connection preface.
//
// The client connection preface is unambiguous on any port, and must be
// checked before the detectors below: is_redis_query takes anything
// starting with an uppercase letter, "PRI * HTTP/2.0" included. On
// other ports the connection was cached as Redis, and every write
// until the first server frame was read (the preface and, from Go and
// gRPC clients, the first request's headers) was lost. Those headers
// carry most of the HPACK dynamic table's insertions, so the table
// was missing them for the life of the connection.
if (conn->dport != 53 && http2_detection_allowed(conn) &&
(is_http2_client_preface(payload, size) ||
(is_likely_http2_port(conn->dport) && looks_like_http2_frame(payload, size, METHOD_HTTP2_CLIENT_FRAMES)))) {
conn->protocol = PROTOCOL_HTTP2; // Cache for subsequent frames
struct l7_event *e = reserve_l7_event();
if (!e) { return 0; }
Expand Down
103 changes: 98 additions & 5 deletions ebpftracer/elf.go
Original file line number Diff line number Diff line change
Expand Up @@ -4,6 +4,7 @@ import (
"debug/elf"
"fmt"
"io"
"os"

"github.com/cilium/ebpf"
"github.com/cilium/ebpf/link"
Expand Down Expand Up @@ -70,6 +71,31 @@ func (s *Symbol) ReturnOffsets() ([]int, error) {
return offsets, nil
}

// StackCheckEnd returns the offset of the first instruction past the
// function's stack check, or 0 if its prologue has none that is recognized.
// See stackCheckEnd.
func (s *Symbol) StackCheckEnd() (int, error) {
text, reader, err := s.f.getTextSectionAndReader()
if err != nil {
return 0, err
}
if s.value < text.Addr || s.size > text.Size || s.value-text.Addr > text.Size-s.size {
return 0, fmt.Errorf("symbol %s [%#x, +%d) is outside .text", s.name, s.value, s.size)
}
n := s.size
if n > stackCheckWindow {
n = stackCheckWindow
}
if _, err := reader.Seek(int64(s.value-text.Addr), io.SeekStart); err != nil {
return 0, err
}
b := make([]byte, n)
if _, err := io.ReadFull(reader, b); err != nil {
return 0, err
}
return stackCheckEnd(s.f.elf.Machine, b), nil
}

func (s *Symbol) AttachUprobe(exe *link.Executable, prog *ebpf.Program, pid uint32) (link.Link, error) {
return exe.Uprobe("", prog, &link.UprobeOptions{Address: s.Address(), PID: int(pid)})
}
Expand All @@ -91,7 +117,10 @@ func (s *Symbol) AttachUretprobes(exe *link.Executable, prog *ebpf.Program, pid
}

type ELFFile struct {
path string
path string
// file is the open binary. Everything is read through it: reopening path,
// a /proc/<pid>/exe link, fails once that process has exited.
file *os.File
elf *elf.File
symbols []elf.Symbol
textSection *elf.Section
Expand All @@ -101,11 +130,16 @@ type ELFFile struct {
}

func OpenELFFile(path string) (*ELFFile, error) {
file, err := elf.Open(path)
file, err := os.Open(path)
if err != nil {
return nil, err
}
return &ELFFile{path: path, elf: file}, nil
ef, err := elf.NewFile(file)
if err != nil {
file.Close()
return nil, err
}
return &ELFFile{path: path, file: file, elf: ef}, nil
}

func (f *ELFFile) readSymbols() error {
Expand Down Expand Up @@ -151,7 +185,7 @@ func (f *ELFFile) GetSymbol(name string) (*Symbol, error) {
// cannot be read.
func (f *ELFFile) goFuncTable() *goFuncTable {
if f.goFuncs == nil && f.goFuncsErr == nil {
f.goFuncs, f.goFuncsErr = openGoFuncTable(f.path, f.elf)
f.goFuncs, f.goFuncsErr = openGoFuncTable(f.file, f.elf)
}
return f.goFuncs
}
Expand All @@ -171,7 +205,66 @@ func (f *ELFFile) Close() error {
if f.goFuncs != nil {
f.goFuncs.close()
}
return f.elf.Close()
// elf.File.Close does nothing for a file made with elf.NewFile.
return f.file.Close()
}

// stackCheckWindow bounds the prologue scanned for the stack check: at most
// four instructions precede its branch.
const stackCheckWindow = 32

// stackCheckEnd returns the offset just past a Go function's stack check: the
// compare against the goroutine's stackguard0 and the branch to morestack
// that follows it. 0 if the prologue does not have that shape.
//
// When the stack has to grow, morestack copies it and restarts the function
// from its first instruction, so a probe at the entry fires twice for one
// call; a probe past the branch runs once the check has passed, exactly once
// per call. The argument registers are untouched up to there: the check only
// uses scratch registers (R12 on amd64, R16/R17 on arm64).
func stackCheckEnd(machine elf.Machine, instructions []byte) int {
switch machine {
case elf.EM_X86_64:
// CMPQ SP, 16(R14) or LEAQ -n(SP), R12; CMPQ R12, 16(R14), then JBE.
compared := false
for i, k := 0, 0; i < len(instructions) && k < 4; k++ {
ins, err := x86asm.Decode(instructions[i:], 64)
if err != nil {
return 0
}
i += ins.Len
switch {
case ins.Op == x86asm.CMP:
for _, a := range ins.Args {
if m, ok := a.(x86asm.Mem); ok && m.Base == x86asm.R14 && m.Disp == 16 {
compared = true
}
}
case ins.Op == x86asm.JBE && compared:
return i
}
}
case elf.EM_AARCH64:
// MOVD 16(g), R16; [SUB $n, RSP, R17;] CMP; BLS.
loaded := false
for i, k := 0, 0; i+4 <= len(instructions) && k < 4; i, k = i+4, k+1 {
ins, err := arm64asm.Decode(instructions[i:])
if err != nil {
return 0
}
switch {
case ins.Op == arm64asm.LDR:
if m, ok := ins.Args[1].(arm64asm.MemImmediate); ok && m.Base == arm64asm.RegSP(arm64asm.X28) {
loaded = true
}
case ins.Op == arm64asm.B && loaded:
if c, ok := ins.Args[0].(arm64asm.Cond); ok && c.Value == 9 { // LS
return i + 4
}
}
}
}
return 0
}

func getReturnOffsets(machine elf.Machine, instructions []byte) []int {
Expand Down
8 changes: 2 additions & 6 deletions ebpftracer/gopclntab.go
Original file line number Diff line number Diff line change
Expand Up @@ -39,7 +39,7 @@ type goFuncTable struct {
textEnd uint64 // end of .text; a function must lie in [textStart, textEnd)
}

func openGoFuncTable(path string, ef *elf.File) (*goFuncTable, error) {
func openGoFuncTable(file *os.File, ef *elf.File) (*goFuncTable, error) {
sec := ef.Section(".gopclntab")
if sec == nil {
// PIE binaries place it in the relocated read-only data.
Expand All @@ -49,11 +49,7 @@ func openGoFuncTable(path string, ef *elf.File) (*goFuncTable, error) {
return nil, errNoGoFuncTable
}

file, err := os.Open(path)
if err != nil {
return nil, err
}
defer file.Close()
// The mapping outlives file: closing the descriptor does not unmap it.
info, err := file.Stat()
if err != nil {
return nil, err
Expand Down
Loading
Loading